mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-05 02:43:32 +00:00
Merge branch 'main' into codex/cuda-cpp-extensions
This commit is contained in:
commit
414c3a0541
30 changed files with 3565 additions and 652 deletions
18
.github/workflows/ci-tests.yml
vendored
18
.github/workflows/ci-tests.yml
vendored
|
|
@ -276,6 +276,24 @@ jobs:
|
|||
run: node --expose-gc --import tsx bench/cfg/measure.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Emit-persistence throughput / byte-identity guards (#2203)
|
||||
# Build-free: asserts streamAllCSVsToDisk output is byte-identical
|
||||
# (order-independent CSV-line fingerprint — the #2203 U2/U3 emit
|
||||
# optimisations must not change graph content) and that emit wall-time
|
||||
# stays linear in node+edge count. The LadybugDB COPY half needs a real
|
||||
# DB, so its timing lives in the runtime PROF_LBUG_LOAD breakdown.
|
||||
run: node --import tsx bench/emit-persistence/measure.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Streaming PDG-emit byte-identity / bounded-RSS guards (#2202)
|
||||
# Build-free: asserts the streaming PdgEmitSink emits a CSV row SET
|
||||
# byte-identical to the whole-graph streamAllCSVsToDisk emit, AND that
|
||||
# the in-memory graph retains zero BasicBlock nodes (the O(chunk) peak-RSS
|
||||
# bound that unblocks full-kernel-scale repos). Fails on fingerprint drift
|
||||
# or any resident BasicBlock.
|
||||
run: node --import tsx bench/emit-persistence/measure-streaming.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial)
|
||||
env:
|
||||
GITNEXUS_BENCH: '1'
|
||||
|
|
|
|||
|
|
@ -324,6 +324,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max
|
|||
| `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. |
|
||||
| `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. |
|
||||
| `GITNEXUS_PROFILE_DEFERRED_SLOW_MS` | `3000` (verbose) / `5000` | Per-file threshold in ms above which `processCallsFromExtracted` emits a `slow file …` log line. Parsed via `Number()`: accepts integers (`5000`), scientific notation (`2.5e3`), decimals (`.5`), and hex (`0x10`). Non-finite or non-positive values fall back to the default. | Hunting a few outlier files dominating the deferred call-resolution stage; lower to surface more, raise to focus only on the worst. |
|
||||
| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. |
|
||||
| `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size <kb>`. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. |
|
||||
| `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout <seconds>` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. |
|
||||
| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold <bytes>`. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. |
|
||||
|
|
|
|||
58
gitnexus/bench/emit-persistence/README.md
Normal file
58
gitnexus/bench/emit-persistence/README.md
Normal file
|
|
@ -0,0 +1,58 @@
|
|||
# Emit-persistence bench (#2203)
|
||||
|
||||
Build-free throughput + byte-identity guard for the **CSV-generation half** of
|
||||
the graph-DB persistence pipeline (`streamAllCSVsToDisk`), which dominates
|
||||
large-repo `analyze` wall time alongside parsing (issue #2203).
|
||||
|
||||
```bash
|
||||
# from gitnexus/
|
||||
node --import tsx bench/emit-persistence/measure.mjs # print one JSON line
|
||||
node --import tsx bench/emit-persistence/measure.mjs --check # gate vs baselines.json
|
||||
```
|
||||
|
||||
## What it measures
|
||||
|
||||
A synthetic `KnowledgeGraph` (files + functions + classes + 4 edge types across
|
||||
the `File→Function`, `File→Class`, `Function→Function` label pairs) at two
|
||||
scales:
|
||||
|
||||
- **`elapsed_ms_small` / `elapsed_ms_large`** — median wall-clock over `REPS`
|
||||
runs of `streamAllCSVsToDisk`.
|
||||
- **`scaling_ratio`** — `(t_large/t_small)/(LARGE/SMALL)`; ~1.0 is linear. The
|
||||
`--check` gate fails if it exceeds `scaling_budget` (catches an O(n²)
|
||||
re-regression in the emit/routing path).
|
||||
- **`fingerprint`** — order-independent sha256 over every emitted CSV line (node
|
||||
CSVs + per-FROM→TO-label-pair rel CSVs). This is the **byte-identity gate**:
|
||||
the U2 (direct per-pair routing) and U3 (per-row microtask elimination)
|
||||
optimisations must not change graph content, and any future change that does
|
||||
fails `--check`. Byte-identity holds for all quote-free ids; for an id
|
||||
containing a `"` the router intentionally diverges from — and is more correct
|
||||
than — the legacy regex oracle (see `src/core/lbug/rel-pair-routing.ts`).
|
||||
|
||||
## What it does NOT measure
|
||||
|
||||
- **The LadybugDB `COPY` half.** Bulk loading needs a live writable DB
|
||||
connection, so it can't run build-free. Its per-stage timing lives in the
|
||||
runtime `PROF_LBUG_LOAD=1` breakdown (`[lbug-load prof] csv-emit=… copy-nodes=…
|
||||
copy-rels=… fallback=… total=…`) and is exercised end-to-end by the
|
||||
integration round-trip tests (`test/integration/basicblock-roundtrip.test.ts`,
|
||||
`lbug-core-adapter.test.ts`).
|
||||
- **Content extraction.** Bench nodes have no backing source files, so the
|
||||
`content` column is empty — emit cost here reflects the CSV machinery
|
||||
(routing, escaping, buffering, disk writes), not file reads.
|
||||
- **At-scale absolute numbers.** The real postgres / kernel-`fs/` wall (issue
|
||||
#2203's table) is a maintainer-run measurement; this synthetic bench is the
|
||||
reproducible regression guard, not a substitute for those runs.
|
||||
|
||||
## Deferred follow-up
|
||||
|
||||
Parallelising the `COPY` loop (`PARALLEL=false` is load-bearing; LadybugDB is
|
||||
single-writer) is **out of scope** for #2203 pending empirical validation of
|
||||
concurrent-COPY support — the `PROF_LBUG_LOAD` breakdown is the prerequisite
|
||||
that shows whether COPY is the dominant cost worth that risk.
|
||||
|
||||
## Regenerating the baseline
|
||||
|
||||
```bash
|
||||
node --import tsx bench/emit-persistence/measure.mjs # copy fingerprint + ratio into baselines.json
|
||||
```
|
||||
4
gitnexus/bench/emit-persistence/baselines-streaming.json
Normal file
4
gitnexus/bench/emit-persistence/baselines-streaming.json
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
{
|
||||
"fingerprint": "386f432c74f4992455055d8891dbe6c873ea95afa60ef4be023a21aed7b4bcb1",
|
||||
"_note": "Byte-identity + bounded-retention gate for streaming/chunked PDG emit (#2202). fingerprint = sha256 of the sorted, header-stripped BasicBlock + PDG-edge data rows of the canonical synthetic set. --check also asserts the streamed PdgEmitSink output is byte-identical to the whole-graph streamAllCSVsToDisk emit (byte_identical_nodes/edges) and that the in-memory graph retains 0 BasicBlocks (resident_basic_blocks === 0, the O(chunk) RSS bound). Regenerate via `node --import tsx bench/emit-persistence/measure-streaming.mjs`."
|
||||
}
|
||||
6
gitnexus/bench/emit-persistence/baselines.json
Normal file
6
gitnexus/bench/emit-persistence/baselines.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"fingerprint": "1b9dd0b783899b47067c36511d241860f291ac736e57682b0ece14148e3958ff",
|
||||
"scaling_budget": 1.8,
|
||||
"max_ms_large": 1000,
|
||||
"_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted — binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`."
|
||||
}
|
||||
199
gitnexus/bench/emit-persistence/measure-streaming.mjs
Normal file
199
gitnexus/bench/emit-persistence/measure-streaming.mjs
Normal file
|
|
@ -0,0 +1,199 @@
|
|||
/**
|
||||
* Build-free byte-identity + bounded-retention bench for streaming/chunked PDG
|
||||
* graph emit (issue #2202).
|
||||
*
|
||||
* Proves the two acceptance criteria at scale, without a DB connection:
|
||||
* 1. BYTE-IDENTITY (R2): emitting a BasicBlock + intra-file PDG-edge set via
|
||||
* the streaming `PdgEmitSink` produces the IDENTICAL CSV data-row set as
|
||||
* the whole-graph `streamAllCSVsToDisk` path. Compared per file
|
||||
* (basicblock.csv, rel_BasicBlock_BasicBlock.csv) over header-stripped,
|
||||
* sorted lines so it is a pure function of the emitted row SET.
|
||||
* 2. BOUNDED RETENTION (R1): with streaming on, the in-memory graph holds
|
||||
* ZERO BasicBlock nodes regardless of how many are emitted — the PDG layer
|
||||
* never accumulates in process memory (peak RSS O(chunk), not O(graph)).
|
||||
*
|
||||
* Build-free: imports the `.ts` hotpaths through tsx
|
||||
* (`node --import tsx bench/emit-persistence/measure-streaming.mjs`).
|
||||
*
|
||||
* Without args: prints one JSON object. With `--check`: asserts byte-identity,
|
||||
* retention, and fingerprint == the committed baseline; exits non-zero on any
|
||||
* failure.
|
||||
*/
|
||||
import fs from 'node:fs';
|
||||
import fsp from 'node:fs/promises';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
import crypto from 'node:crypto';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.ts';
|
||||
import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.ts';
|
||||
import { PdgEmitSink } from '../../src/core/lbug/pdg-emit-sink.ts';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const BASELINE_PATH = path.resolve(__dirname, 'baselines-streaming.json');
|
||||
|
||||
// PDG edge types streamed per file (all intra-block BasicBlock→BasicBlock).
|
||||
const PDG_TYPES = ['CFG', 'REACHING_DEF', 'CDG', 'POST_DOMINATE', 'TAINTED', 'SANITIZES'];
|
||||
|
||||
const FUNCS = 1200; // functions
|
||||
const BLOCKS = 6; // basic blocks per function ⇒ FUNCS*BLOCKS BasicBlocks total
|
||||
const CHUNK_ROWS = 64; // tiny streamed buffer to exercise frequent flushing
|
||||
|
||||
/**
|
||||
* Build the canonical PDG node/edge SET: `FUNCS` functions each with `BLOCKS`
|
||||
* BasicBlocks and a chain of intra-function PDG edges. Returns the structural
|
||||
* nodes (File/Function) separately from the BasicBlock + PDG-edge layer so the
|
||||
* streamed path can route them to different sinks.
|
||||
*/
|
||||
function buildSet() {
|
||||
const structuralNodes = [];
|
||||
const structuralRels = [];
|
||||
const bbNodes = [];
|
||||
const pdgEdges = [];
|
||||
for (let f = 0; f < FUNCS; f++) {
|
||||
const fp = `src/m${f % 50}.ts`;
|
||||
const fnId = `Function:${fp}:fn${f}:1`;
|
||||
structuralNodes.push({
|
||||
id: fnId,
|
||||
label: 'Function',
|
||||
properties: { name: `fn${f}`, filePath: fp, startLine: 1, endLine: 99 },
|
||||
});
|
||||
for (let b = 0; b < BLOCKS; b++) {
|
||||
bbNodes.push({
|
||||
id: `BasicBlock:${fp}:1:0:${f}_${b}`,
|
||||
label: 'BasicBlock',
|
||||
properties: {
|
||||
name: '',
|
||||
filePath: fp,
|
||||
startLine: b * 3,
|
||||
endLine: b * 3 + 2,
|
||||
text: `f${f}b${b}`,
|
||||
},
|
||||
});
|
||||
}
|
||||
for (let b = 0; b < BLOCKS - 1; b++) {
|
||||
const from = `BasicBlock:${fp}:1:0:${f}_${b}`;
|
||||
const to = `BasicBlock:${fp}:1:0:${f}_${b + 1}`;
|
||||
for (const type of PDG_TYPES) {
|
||||
pdgEdges.push({
|
||||
id: `${type}:${f}:${b}`,
|
||||
sourceId: from,
|
||||
targetId: to,
|
||||
type,
|
||||
confidence: 1,
|
||||
reason: type === 'REACHING_DEF' ? `v${b}` : type === 'CDG' ? 'T' : '',
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
// A few File nodes so the structural emit produces a realistic multi-table mix.
|
||||
for (let m = 0; m < 50; m++) {
|
||||
structuralNodes.push({
|
||||
id: `File:src/m${m}.ts`,
|
||||
label: 'File',
|
||||
properties: { name: `m${m}.ts`, filePath: `src/m${m}.ts` },
|
||||
});
|
||||
}
|
||||
return { structuralNodes, structuralRels, bbNodes, pdgEdges };
|
||||
}
|
||||
|
||||
/** Header-stripped, sorted, non-empty data rows of one CSV file (or [] if absent). */
|
||||
async function dataRows(csvPath) {
|
||||
let text;
|
||||
try {
|
||||
text = await fsp.readFile(csvPath, 'utf8');
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
const lines = text.split('\n').filter((l) => l.length > 0);
|
||||
return lines.slice(1).sort(); // drop the header line
|
||||
}
|
||||
|
||||
const sha = (rows) => crypto.createHash('sha256').update(rows.join('\n')).digest('hex');
|
||||
|
||||
async function measure() {
|
||||
// mkdtemp (unpredictable, unique) rather than a predictable pid-based tmp path.
|
||||
const tmpRoot = await fsp.mkdtemp(path.join(os.tmpdir(), 'gitnexus-stream-bench-'));
|
||||
try {
|
||||
const { structuralNodes, structuralRels, bbNodes, pdgEdges } = buildSet();
|
||||
|
||||
// ── whole-graph path ─────────────────────────────────────────────────
|
||||
const wholeGraph = createKnowledgeGraph();
|
||||
for (const n of structuralNodes) wholeGraph.addNode(n);
|
||||
for (const n of bbNodes) wholeGraph.addNode(n);
|
||||
for (const r of structuralRels) wholeGraph.addRelationship(r);
|
||||
for (const e of pdgEdges) wholeGraph.addRelationship(e);
|
||||
const wholeDir = path.join(tmpRoot, 'whole');
|
||||
await streamAllCSVsToDisk(wholeGraph, path.join(tmpRoot, 'no-repo'), wholeDir);
|
||||
|
||||
// ── streamed path ────────────────────────────────────────────────────
|
||||
const realGraph = createKnowledgeGraph();
|
||||
const sink = new PdgEmitSink(realGraph, path.join(tmpRoot, 'pdg-csv'), CHUNK_ROWS);
|
||||
for (const n of structuralNodes) realGraph.addNode(n); // structural → real graph
|
||||
for (const r of structuralRels) realGraph.addRelationship(r);
|
||||
for (const n of bbNodes) sink.addNode(n); // BasicBlock layer → sink (CSV)
|
||||
for (const e of pdgEdges) sink.addRelationship(e);
|
||||
sink.finalize();
|
||||
const streamedCsvDir = path.join(tmpRoot, 'streamed');
|
||||
await streamAllCSVsToDisk(realGraph, path.join(tmpRoot, 'no-repo'), streamedCsvDir);
|
||||
|
||||
// ── retention (R1): the real graph holds ZERO BasicBlocks ────────────
|
||||
let residentBasicBlocks = 0;
|
||||
for (const n of realGraph.iterNodes()) if (n.label === 'BasicBlock') residentBasicBlocks++;
|
||||
|
||||
// ── byte-identity (R2): per-file data-row set equality ───────────────
|
||||
const wholeBb = await dataRows(path.join(wholeDir, 'basicblock.csv'));
|
||||
const streamedBb = await dataRows(path.join(tmpRoot, 'pdg-csv', 'basicblock.csv'));
|
||||
const wholeRel = await dataRows(path.join(wholeDir, 'rel_BasicBlock_BasicBlock.csv'));
|
||||
const streamedRel = await dataRows(
|
||||
path.join(tmpRoot, 'pdg-csv', 'rel_BasicBlock_BasicBlock.csv'),
|
||||
);
|
||||
|
||||
const bbIdentical = sha(wholeBb) === sha(streamedBb);
|
||||
const relIdentical = sha(wholeRel) === sha(streamedRel);
|
||||
// Fingerprint over the canonical PDG data-row set (drift gate).
|
||||
const fingerprint = sha([...wholeBb, ...wholeRel].sort());
|
||||
|
||||
return {
|
||||
scenario: 'streamingPdgEmit',
|
||||
basic_blocks: bbNodes.length,
|
||||
pdg_edges: pdgEdges.length,
|
||||
chunk_rows: CHUNK_ROWS,
|
||||
resident_basic_blocks: residentBasicBlocks,
|
||||
byte_identical_nodes: bbIdentical,
|
||||
byte_identical_edges: relIdentical,
|
||||
fingerprint,
|
||||
};
|
||||
} finally {
|
||||
await fsp.rm(tmpRoot, { recursive: true, force: true }).catch(() => {});
|
||||
}
|
||||
}
|
||||
|
||||
const CHECK = process.argv.includes('--check');
|
||||
const result = await measure();
|
||||
|
||||
if (!CHECK) {
|
||||
process.stdout.write(JSON.stringify(result) + '\n');
|
||||
} else {
|
||||
const base = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8'));
|
||||
const failures = [];
|
||||
if (!result.byte_identical_nodes)
|
||||
failures.push('streamed BasicBlock rows differ from whole-graph emit');
|
||||
if (!result.byte_identical_edges)
|
||||
failures.push('streamed PDG-edge rows differ from whole-graph emit');
|
||||
if (result.resident_basic_blocks !== 0) {
|
||||
failures.push(
|
||||
`RSS bound violated: ${result.resident_basic_blocks} BasicBlock node(s) retained in the in-memory graph (expected 0)`,
|
||||
);
|
||||
}
|
||||
if (result.fingerprint !== base.fingerprint) {
|
||||
failures.push(`fingerprint drift (got ${result.fingerprint}, expected ${base.fingerprint})`);
|
||||
}
|
||||
process.stdout.write(JSON.stringify(result) + '\n');
|
||||
if (failures.length > 0) {
|
||||
for (const f of failures) process.stderr.write(`[stream-pdg-emit --check] FAIL: ${f}\n`);
|
||||
process.exit(1);
|
||||
}
|
||||
process.stderr.write('[stream-pdg-emit --check] PASS\n');
|
||||
}
|
||||
216
gitnexus/bench/emit-persistence/measure.mjs
Normal file
216
gitnexus/bench/emit-persistence/measure.mjs
Normal file
|
|
@ -0,0 +1,216 @@
|
|||
/**
|
||||
* Build-free emit-path throughput + byte-identity bench for the graph-DB
|
||||
* persistence pipeline (issue #2203).
|
||||
*
|
||||
* Measures `streamAllCSVsToDisk` — the CSV-generation half of the persistence
|
||||
* path that U2 (direct per-pair relationship routing) and U3 (per-row
|
||||
* microtask elimination) optimised. The LadybugDB `COPY` half needs a real DB
|
||||
* connection, so its timing lives in the runtime `PROF_LBUG_LOAD` breakdown +
|
||||
* the integration round-trip tests, NOT here (see README.md).
|
||||
*
|
||||
* For a synthetic KnowledgeGraph at two scales it reports:
|
||||
* - elapsed_ms_small / elapsed_ms_large (median over REPS) + a scaling ratio
|
||||
* `(t_large/t_small)/(LARGE/SMALL)`: ~1.0 linear, ~3.x quadratic;
|
||||
* - an order-independent sha256 fingerprint over every emitted CSV line
|
||||
* (node CSVs + per-FROM→TO-label-pair rel CSVs), as the byte-identity gate
|
||||
* guarding the issue's "byte-identical graph content" requirement.
|
||||
*
|
||||
* Build-free: imports the `.ts` hotpaths through tsx
|
||||
* (`node --import tsx bench/emit-persistence/measure.mjs`). Static `.ts`
|
||||
* imports work; a top-level `await import()` breaks tsx's lexer.
|
||||
*
|
||||
* Without args: prints one JSON object per scenario.
|
||||
* With `--check`: asserts the fingerprint == the committed baseline AND the
|
||||
* scaling ratio < the recorded budget; exits non-zero on drift/regression.
|
||||
*/
|
||||
import fs from 'node:fs';
|
||||
import fsp from 'node:fs/promises';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
import crypto from 'node:crypto';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.ts';
|
||||
import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.ts';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const BASELINE_PATH = path.resolve(__dirname, 'baselines.json');
|
||||
|
||||
// ---- synthetic graph generation (deterministic — no randomness) ----
|
||||
|
||||
/**
|
||||
* Build a graph of `entityCount` files, each with 2 functions + 1 class and 4
|
||||
* relationships. ids carry valid table-label prefixes (File:/Function:/Class:)
|
||||
* so edges route to real pairs (File→Function, File→Class, Function→Function)
|
||||
* — exercising the U2 router across multiple label pairs. Node `content` is
|
||||
* never populated (no backing files), so emit cost reflects the CSV machinery
|
||||
* (routing, escaping, buffering, disk writes), not content extraction.
|
||||
*/
|
||||
function generateGraph(entityCount) {
|
||||
const graph = createKnowledgeGraph();
|
||||
for (let i = 0; i < entityCount; i++) {
|
||||
const fp = `src/e${i}.ts`;
|
||||
const fileId = `File:${fp}`;
|
||||
const fnA = `Function:${fp}:fnA:1`;
|
||||
const fnB = `Function:${fp}:fnB:10`;
|
||||
const cls = `Class:${fp}:C:20`;
|
||||
graph.addNode({ id: fileId, label: 'File', properties: { name: `e${i}.ts`, filePath: fp } });
|
||||
graph.addNode({
|
||||
id: fnA,
|
||||
label: 'Function',
|
||||
properties: { name: 'fnA', filePath: fp, startLine: 1, endLine: 5, isExported: true },
|
||||
});
|
||||
graph.addNode({
|
||||
id: fnB,
|
||||
label: 'Function',
|
||||
properties: { name: 'fnB', filePath: fp, startLine: 10, endLine: 15, isExported: false },
|
||||
});
|
||||
graph.addNode({
|
||||
id: cls,
|
||||
label: 'Class',
|
||||
properties: { name: 'C', filePath: fp, startLine: 20, endLine: 30, isExported: true },
|
||||
});
|
||||
graph.addRelationship({
|
||||
id: `${fileId}->${fnA}`,
|
||||
sourceId: fileId,
|
||||
targetId: fnA,
|
||||
type: 'CONTAINS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
graph.addRelationship({
|
||||
id: `${fileId}->${fnB}`,
|
||||
sourceId: fileId,
|
||||
targetId: fnB,
|
||||
type: 'CONTAINS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
graph.addRelationship({
|
||||
id: `${fileId}->${cls}`,
|
||||
sourceId: fileId,
|
||||
targetId: cls,
|
||||
type: 'CONTAINS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
graph.addRelationship({
|
||||
id: `${fnA}->${fnB}`,
|
||||
sourceId: fnA,
|
||||
targetId: fnB,
|
||||
type: 'CALLS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
}
|
||||
return graph;
|
||||
}
|
||||
|
||||
// ---- byte-identity fingerprint (order-independent) ----
|
||||
|
||||
/** sha256 over every non-empty line of every emitted CSV file, sorted so the
|
||||
* digest is a pure function of the emitted line SET (insertion-order agnostic). */
|
||||
async function fingerprintEmit(graph, dir) {
|
||||
await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir);
|
||||
// Per-file digest bound to the filename: a row routed to the WRONG pair file
|
||||
// (or a header written to the wrong file) changes the fingerprint — a global
|
||||
// line-flatten could not catch that. File bytes are hashed as-written (so it
|
||||
// also catches within-file row reordering); the entry list is sorted so
|
||||
// readdir order doesn't matter.
|
||||
const entries = [];
|
||||
for (const name of fs.readdirSync(dir)) {
|
||||
if (!name.endsWith('.csv')) continue;
|
||||
const bytes = await fsp.readFile(path.join(dir, name));
|
||||
entries.push(`${name}\n${crypto.createHash('sha256').update(bytes).digest('hex')}`);
|
||||
}
|
||||
return crypto.createHash('sha256').update(entries.sort().join('\n')).digest('hex');
|
||||
}
|
||||
|
||||
// ---- timing ----
|
||||
|
||||
function median(xs) {
|
||||
const s = [...xs].sort((a, b) => a - b);
|
||||
const m = Math.floor(s.length / 2);
|
||||
return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2;
|
||||
}
|
||||
|
||||
async function timeEmit(graph, dir, reps) {
|
||||
await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); // warmup (not counted)
|
||||
const samples = [];
|
||||
for (let i = 0; i < reps; i++) {
|
||||
const start = process.hrtime.bigint();
|
||||
await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir);
|
||||
samples.push(Number(process.hrtime.bigint() - start) / 1e6);
|
||||
}
|
||||
return median(samples);
|
||||
}
|
||||
|
||||
const SMALL = 600;
|
||||
const LARGE = 2400;
|
||||
const REPS = 5;
|
||||
|
||||
async function measure() {
|
||||
const tmpRoot = path.join(os.tmpdir(), `gitnexus-emit-bench-${process.pid}`);
|
||||
await fsp.mkdir(tmpRoot, { recursive: true });
|
||||
try {
|
||||
const smallGraph = generateGraph(SMALL);
|
||||
const largeGraph = generateGraph(LARGE);
|
||||
|
||||
const fingerprint = await fingerprintEmit(largeGraph, path.join(tmpRoot, 'fp'));
|
||||
const small = await timeEmit(smallGraph, path.join(tmpRoot, 'small'), REPS);
|
||||
const large = await timeEmit(largeGraph, path.join(tmpRoot, 'large'), REPS);
|
||||
const scalingRatio = small > 0 ? large / small / (LARGE / SMALL) : 0;
|
||||
|
||||
return {
|
||||
scenario: 'streamAllCSVsToDisk',
|
||||
entities_small: SMALL,
|
||||
entities_large: LARGE,
|
||||
nodes_large: LARGE * 4,
|
||||
rels_large: LARGE * 4,
|
||||
elapsed_ms_small: Number(small.toFixed(2)),
|
||||
elapsed_ms_large: Number(large.toFixed(2)),
|
||||
scaling_ratio: Number(scalingRatio.toFixed(3)),
|
||||
fingerprint,
|
||||
};
|
||||
} finally {
|
||||
await fsp.rm(tmpRoot, { recursive: true, force: true }).catch(() => {});
|
||||
}
|
||||
}
|
||||
|
||||
// ---- run ----
|
||||
|
||||
const CHECK = process.argv.includes('--check');
|
||||
const result = await measure();
|
||||
|
||||
if (!CHECK) {
|
||||
process.stdout.write(JSON.stringify(result) + '\n');
|
||||
} else {
|
||||
const base = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8'));
|
||||
const failures = [];
|
||||
if (result.fingerprint !== base.fingerprint) {
|
||||
failures.push(
|
||||
`byte-identity fingerprint drift (got ${result.fingerprint}, expected ${base.fingerprint})`,
|
||||
);
|
||||
}
|
||||
if (result.scaling_ratio >= base.scaling_budget) {
|
||||
failures.push(
|
||||
`scaling ratio ${result.scaling_ratio} >= budget ${base.scaling_budget} ` +
|
||||
`(${SMALL}->${LARGE} entities, ms ${result.elapsed_ms_small}->${result.elapsed_ms_large})`,
|
||||
);
|
||||
}
|
||||
// Absolute backstop: the scaling ratio alone passes a uniform Nx slowdown (it
|
||||
// only compares large/small). A generous, host-noise-tolerant ceiling catches
|
||||
// a gross absolute regression. Opt-in (only enforced when max_ms_large is set).
|
||||
if (base.max_ms_large !== undefined && result.elapsed_ms_large >= base.max_ms_large) {
|
||||
failures.push(
|
||||
`absolute wall-time regression: elapsed_ms_large ${result.elapsed_ms_large}ms >= budget ` +
|
||||
`${base.max_ms_large}ms (coarse backstop, not a tight SLA)`,
|
||||
);
|
||||
}
|
||||
process.stdout.write(JSON.stringify(result) + '\n');
|
||||
if (failures.length > 0) {
|
||||
for (const f of failures) process.stderr.write(`[emit-persistence --check] FAIL: ${f}\n`);
|
||||
process.exit(1);
|
||||
}
|
||||
process.stderr.write('[emit-persistence --check] PASS\n');
|
||||
}
|
||||
46
gitnexus/package-lock.json
generated
46
gitnexus/package-lock.json
generated
|
|
@ -1340,19 +1340,18 @@
|
|||
"license": "BSD-3-Clause"
|
||||
},
|
||||
"node_modules/@protobufjs/eventemitter": {
|
||||
"version": "1.1.0",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.0.tgz",
|
||||
"integrity": "sha512-j9ednRT81vYJ9OfVuXG6ERSTdEL1xVsNgqpkxMsbIabzSo3goCjDIveeGv5d03om39ML71RdmrGNjG5SReBP/Q==",
|
||||
"version": "1.1.1",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.1.tgz",
|
||||
"integrity": "sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg==",
|
||||
"license": "BSD-3-Clause"
|
||||
},
|
||||
"node_modules/@protobufjs/fetch": {
|
||||
"version": "1.1.0",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.0.tgz",
|
||||
"integrity": "sha512-lljVXpqXebpsijW71PZaCYeIcE5on1w5DlQy5WH6GLbFryLUrBD4932W/E2BSpfRJWseIL4v/KPgBFxDOIdKpQ==",
|
||||
"version": "1.1.1",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.1.tgz",
|
||||
"integrity": "sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw==",
|
||||
"license": "BSD-3-Clause",
|
||||
"dependencies": {
|
||||
"@protobufjs/aspromise": "^1.1.1",
|
||||
"@protobufjs/inquire": "^1.1.0"
|
||||
"@protobufjs/aspromise": "^1.1.1"
|
||||
}
|
||||
},
|
||||
"node_modules/@protobufjs/float": {
|
||||
|
|
@ -1361,12 +1360,6 @@
|
|||
"integrity": "sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ==",
|
||||
"license": "BSD-3-Clause"
|
||||
},
|
||||
"node_modules/@protobufjs/inquire": {
|
||||
"version": "1.1.1",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/inquire/-/inquire-1.1.1.tgz",
|
||||
"integrity": "sha512-mnzgDV26ueAvk7rsbt9L7bE0SuAoqyuys/sMMrmVcN5x9VsxpcG3rqAUSgDyLp0UZlmNfIbQ4fHfCtreVBk8Ew==",
|
||||
"license": "BSD-3-Clause"
|
||||
},
|
||||
"node_modules/@protobufjs/path": {
|
||||
"version": "1.1.2",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/path/-/path-1.1.2.tgz",
|
||||
|
|
@ -1822,9 +1815,9 @@
|
|||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/node": {
|
||||
"version": "25.9.2",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.2.tgz",
|
||||
"integrity": "sha512-G05zqtJhcDLb8uslf5EjCxXg9G1KQxiV8OS0R26IC//Eoyitzqe8z37I7cqvnZlrlSfgocQRfSn/AHBZJJFyGw==",
|
||||
"version": "25.9.3",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.3.tgz",
|
||||
"integrity": "sha512-603BddQMv3pUcr4U2dhujk83N2tTDVr/34wII2B6bJy6g+8WD6yUb11jszNs0gdi4PesVWl7ABt8nYMVpnLUcg==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"undici-types": ">=7.24.0 <7.24.7"
|
||||
|
|
@ -4351,24 +4344,23 @@
|
|||
"license": "MIT"
|
||||
},
|
||||
"node_modules/protobufjs": {
|
||||
"version": "7.5.8",
|
||||
"resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.5.8.tgz",
|
||||
"integrity": "sha512-dvpCIeLPbXZS/Ete7yLaO7RenOdken2NHKykBXbsaGxZT0UTltcarBciw+A78SRQs9iMAAVpsYA+l8b1hTePIA==",
|
||||
"version": "7.6.4",
|
||||
"resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.6.4.tgz",
|
||||
"integrity": "sha512-RJJPTTpvFfHcWLkIa2JFWK4XvtSzS0yEWDmunqHXli1h3JlkbcQZXDZdcWxv+JK3Xsl5/UFDPZ0iGm7DAengYw==",
|
||||
"hasInstallScript": true,
|
||||
"license": "BSD-3-Clause",
|
||||
"dependencies": {
|
||||
"@protobufjs/aspromise": "^1.1.2",
|
||||
"@protobufjs/base64": "^1.1.2",
|
||||
"@protobufjs/codegen": "^2.0.5",
|
||||
"@protobufjs/eventemitter": "^1.1.0",
|
||||
"@protobufjs/fetch": "^1.1.0",
|
||||
"@protobufjs/eventemitter": "^1.1.1",
|
||||
"@protobufjs/fetch": "^1.1.1",
|
||||
"@protobufjs/float": "^1.0.2",
|
||||
"@protobufjs/inquire": "^1.1.1",
|
||||
"@protobufjs/path": "^1.1.2",
|
||||
"@protobufjs/pool": "^1.1.0",
|
||||
"@protobufjs/utf8": "^1.1.1",
|
||||
"@types/node": ">=13.7.0",
|
||||
"long": "^5.0.0"
|
||||
"long": "^5.3.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=12.0.0"
|
||||
|
|
@ -4917,9 +4909,9 @@
|
|||
}
|
||||
},
|
||||
"node_modules/tar": {
|
||||
"version": "7.5.13",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.13.tgz",
|
||||
"integrity": "sha512-tOG/7GyXpFevhXVh8jOPJrmtRpOTsYqUIkVdVooZYJS/z8WhfQUX8RJILmeuJNinGAMSu1veBr4asSHFt5/hng==",
|
||||
"version": "7.5.16",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz",
|
||||
"integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"dependencies": {
|
||||
"@isaacs/fs-minipass": "^4.0.0",
|
||||
|
|
|
|||
|
|
@ -121,6 +121,23 @@ export interface PipelineOptions {
|
|||
/** Per-run `TAINT_PATH` edge cap (#2084 review P1-3). `undefined` ⇒
|
||||
* `DEFAULT_PDG_MAX_INTERPROC_EDGES` (1000); `0` ⇒ no cap. */
|
||||
pdgMaxInterprocEdges?: number;
|
||||
/**
|
||||
* Streaming/chunked PDG graph emit (#2202). When true, the BasicBlock +
|
||||
* intra-file PDG-edge layer (CFG / REACHING_DEF / CDG / POST_DOMINATE /
|
||||
* TAINTED / SANITIZES) is streamed to CSV-on-disk during the scope-resolution
|
||||
* emit loop instead of being materialized in the in-memory graph, bounding
|
||||
* peak RSS to O(chunk) rather than O(graph) at full-kernel scale. Already
|
||||
* gated by the caller to full-rebuild runs only (the incremental writeback
|
||||
* reads BasicBlocks back from the in-memory graph). Memory-only — produces a
|
||||
* byte-identical persisted graph and is NOT part of `RepoMeta.pdg`, so
|
||||
* toggling it never trips `pdgModeMismatch`. Default/false ⇒ today's
|
||||
* whole-graph emit.
|
||||
*/
|
||||
streamPdgEmit?: boolean;
|
||||
/** Streamed PDG-emit write buffer (rows) when `streamPdgEmit` is on (#2202).
|
||||
* `undefined` ⇒ `DEFAULT_PDG_EMIT_CHUNK_ROWS`. Memory-only; does not affect
|
||||
* emitted bytes. */
|
||||
pdgEmitChunkSize?: number;
|
||||
/**
|
||||
* Request parsing with the worker pool disabled. The sequential parser was
|
||||
* removed — the worker pool is the sole parse path — so setting this now
|
||||
|
|
@ -287,10 +304,10 @@ export const runPipelineFromRepo = async (
|
|||
|
||||
let communityResult: CommunitiesOutput['communityResult'] | undefined;
|
||||
let processResult: ProcessesOutput['processResult'] | undefined;
|
||||
const resolutionOutcomes = getPhaseOutput<ScopeResolutionOutput>(
|
||||
results,
|
||||
'scopeResolution',
|
||||
).resolutionOutcomes;
|
||||
const scopeResolutionOutput = getPhaseOutput<ScopeResolutionOutput>(results, 'scopeResolution');
|
||||
const resolutionOutcomes = scopeResolutionOutput.resolutionOutcomes;
|
||||
// Streamed PDG-emit manifest (#2202): present only when streaming was on.
|
||||
const pdgEmitManifest = scopeResolutionOutput.pdgEmitManifest;
|
||||
|
||||
if (!options?.skipGraphPhases) {
|
||||
communityResult = getPhaseOutput<CommunitiesOutput>(results, 'communities').communityResult;
|
||||
|
|
@ -319,5 +336,6 @@ export const runPipelineFromRepo = async (
|
|||
processResult,
|
||||
resolutionOutcomes,
|
||||
usedWorkerPool,
|
||||
pdgEmitManifest,
|
||||
};
|
||||
};
|
||||
|
|
|
|||
|
|
@ -44,6 +44,8 @@ import {
|
|||
import type { ResolutionOutcome } from '../resolution-outcome.js';
|
||||
import type { FunctionSummary } from '../../taint/summary-model.js';
|
||||
import { buildFunctionNodeIndex } from '../../taint/summary-harvest-driver.js';
|
||||
import { PdgEmitSink, type PdgEmitManifest } from '../../../lbug/pdg-emit-sink.js';
|
||||
import { resolveNativeSafeStorageDir } from '../../../lbug/lbug-config.js';
|
||||
|
||||
import { logger } from '../../../logger.js';
|
||||
export interface ScopeResolutionOutput {
|
||||
|
|
@ -72,6 +74,14 @@ export interface ScopeResolutionOutput {
|
|||
* The `taintSummaries` phase composes these over the `CALLS` graph.
|
||||
*/
|
||||
readonly functionSummaries: readonly FunctionSummary[];
|
||||
/**
|
||||
* Streamed PDG-emit COPY manifest (#2202). Present only when streaming was on
|
||||
* (full rebuild + `--pdg` + enabled): the BasicBlock node CSV + per-pair PDG
|
||||
* edge CSVs that were flushed to disk during the emit loop, for the persistence
|
||||
* step to COPY alongside the structural CSVs. Absent ⇒ the PDG layer (if any)
|
||||
* is in the in-memory graph and persists via the normal whole-graph emit.
|
||||
*/
|
||||
readonly pdgEmitManifest?: PdgEmitManifest;
|
||||
}
|
||||
|
||||
const NOOP_OUTPUT: ScopeResolutionOutput = Object.freeze({
|
||||
|
|
@ -242,236 +252,291 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
? buildFunctionNodeIndex(ctx.graph)
|
||||
: undefined;
|
||||
|
||||
for (const [lang, provider] of SCOPE_RESOLVERS) {
|
||||
// Standalone providers (COBOL, JCL) don't emit graph edges yet
|
||||
// through the scope-resolution path. This is the canonical guard:
|
||||
// runScopeResolution is never called for standalone providers, which
|
||||
// keeps cobolPhase as the sole IMPORTS edge producer. Keep this guard
|
||||
// in sync with any additional standalone providers added to
|
||||
// SCOPE_RESOLVERS.
|
||||
if (provider.languageProvider.parseStrategy === 'standalone') continue;
|
||||
|
||||
const primaryLangFiles = filesByLang.get(lang) ?? [];
|
||||
if (primaryLangFiles.length === 0) continue;
|
||||
const primaryFilePaths = primaryLangFiles.map((f) => f.path);
|
||||
|
||||
// Load per-language import-resolution config (tsconfig paths,
|
||||
// composer.json autoload, go.mod, ...). One I/O round trip per
|
||||
// workspace pass — cached implicitly by the result handed to
|
||||
// every `resolveImportTarget` call below.
|
||||
const resolutionConfig =
|
||||
provider.loadResolutionConfig !== undefined
|
||||
? await provider.loadResolutionConfig(ctx.repoPath)
|
||||
: undefined;
|
||||
|
||||
// Some languages (e.g. Vue) expand their file universe beyond the
|
||||
// primary-language files via the `collectScopeContextPaths` hook.
|
||||
// The hook receives raw source contents of the primary files so it
|
||||
// can trace import closures without a second tree-sitter parse.
|
||||
//
|
||||
// To avoid reading primary files twice (once for the hook, once for
|
||||
// the resolution pass), we read them upfront and merge with the
|
||||
// extra context paths the hook may add.
|
||||
// Stream this language's pre-built ParsedFiles in from the disk store
|
||||
// FIRST (huge-repo path). Doing it before reading source lets us skip
|
||||
// loading content for files the store already covers — for a provider
|
||||
// with no content-consuming hook that source is pure dead weight once
|
||||
// extraction is served from the store (~1.5 GB on the kernel's C pass).
|
||||
// Merged into `preExtractedByPath`; the per-language release block below
|
||||
// evicts these again before the next language, so only one language's
|
||||
// ParsedFiles are resident at a time.
|
||||
const loadStoreFor = async (paths: ReadonlySet<string>): Promise<void> => {
|
||||
if (!parsedFileStorePath) return;
|
||||
const fromDisk = await loadParsedFilesForPaths(parsedFileStorePath, paths);
|
||||
for (const [fp, pf] of fromDisk) preExtractedByPath.set(fp, pf);
|
||||
};
|
||||
|
||||
// A provider that feeds source text into a post-extract hook
|
||||
// (populateWorkspaceOwners / populateNamespaceSiblings /
|
||||
// populateRangeBindings / emitPostResolutionEdges) needs content for ALL
|
||||
// its files; one without those hooks only needs content for files the
|
||||
// store does NOT cover (fresh-extract fallback). Keep this in sync with
|
||||
// the getFileContents() call-sites in run.ts.
|
||||
const providerNeedsAllContent =
|
||||
provider.populateWorkspaceOwners !== undefined ||
|
||||
provider.populateNamespaceSiblings !== undefined ||
|
||||
provider.populateRangeBindings !== undefined ||
|
||||
provider.emitPostResolutionEdges !== undefined;
|
||||
|
||||
let scopeFilePaths: Set<string>;
|
||||
let contents: Map<string, string>;
|
||||
if (provider.collectScopeContextPaths !== undefined) {
|
||||
// Context-expanding providers (e.g. Vue) need every primary file's
|
||||
// source up front for the closure hook, so load it all.
|
||||
const entryFileContents = await readFileContents(ctx.repoPath, primaryFilePaths);
|
||||
scopeFilePaths = provider.collectScopeContextPaths({
|
||||
primaryFilePaths,
|
||||
preExtractedByPath,
|
||||
entryFileContents,
|
||||
allScannedPaths,
|
||||
resolutionConfig,
|
||||
});
|
||||
// Read only the extra context files (TS/JS etc.) not already loaded.
|
||||
const extraPaths = [...scopeFilePaths].filter((p) => !entryFileContents.has(p));
|
||||
const extraContents = await readFileContents(ctx.repoPath, extraPaths);
|
||||
contents = new Map([...entryFileContents, ...extraContents]);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
// Streaming/chunked PDG emit (#2202): when enabled (the caller has already
|
||||
// gated this to full-rebuild + `--pdg`), route the BasicBlock + intra-file
|
||||
// PDG-edge layer to CSV-on-disk through one sink shared across every
|
||||
// language pass, so it never accumulates in `ctx.graph` (peak RSS O(chunk)).
|
||||
// Needs the storage dir (the parse-cache store path, the same `.gitnexus`
|
||||
// dir loadGraphToLbug COPYs from); if that is somehow absent we skip
|
||||
// streaming and fall back to the in-memory whole-graph emit.
|
||||
let pdgEmitSink: PdgEmitSink | undefined;
|
||||
if (ctx.options?.streamPdgEmit === true && totalScopeFiles > 0) {
|
||||
if (parsedFileStorePath) {
|
||||
pdgEmitSink = new PdgEmitSink(
|
||||
ctx.graph,
|
||||
// Same ASCII-safe relocation the structural CSVs get (#2202 review #2):
|
||||
// on Windows non-ASCII storage paths the COPY can't open files under
|
||||
// the native path, so the dir is relocated to a hashed os.tmpdir().
|
||||
resolveNativeSafeStorageDir(parsedFileStorePath, 'pdg-csv'),
|
||||
ctx.options?.pdgEmitChunkSize,
|
||||
);
|
||||
} else {
|
||||
scopeFilePaths = new Set(primaryFilePaths);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
const pathsToRead = providerNeedsAllContent
|
||||
? primaryFilePaths
|
||||
: primaryFilePaths.filter((p) => !preExtractedByPath.has(p));
|
||||
contents = await readFileContents(ctx.repoPath, pathsToRead);
|
||||
}
|
||||
const filePaths = [...scopeFilePaths];
|
||||
const files: { path: string; content: string }[] = [];
|
||||
for (const fp of filePaths) {
|
||||
const content = contents.get(fp);
|
||||
if (content !== undefined) {
|
||||
files.push({ path: fp, content });
|
||||
} else if (preExtractedByPath.has(fp)) {
|
||||
// Store covers extraction for this file and we deliberately skipped
|
||||
// reading its source; the empty string is never consumed (the
|
||||
// extract loop uses the pre-extracted ParsedFile and this provider
|
||||
// has no content hook).
|
||||
files.push({ path: fp, content: '' });
|
||||
}
|
||||
// else: uncovered AND unreadable → skip (unchanged from prior behavior).
|
||||
}
|
||||
|
||||
const langFileCount = files.length;
|
||||
logHeapProbe(
|
||||
'scope-lang-start',
|
||||
`lang=${lang} files=${langFileCount} contentsLoaded=${contents.size}`,
|
||||
);
|
||||
const langLabel = lang.charAt(0).toUpperCase() + lang.slice(1);
|
||||
currentLangIdx++;
|
||||
const langTag =
|
||||
totalScopeLangs > 1 ? `${langLabel} [${currentLangIdx}/${totalScopeLangs}]` : langLabel;
|
||||
|
||||
if (totalScopeFiles > 0) {
|
||||
const pct =
|
||||
SCOPE_PCT_START + Math.round((processedScopeFiles / totalScopeFiles) * SCOPE_PCT_RANGE);
|
||||
ctx.onProgress({
|
||||
phase: 'scopeResolution',
|
||||
percent: pct,
|
||||
message: 'Resolving types',
|
||||
detail: `${langTag}, ${langFileCount.toLocaleString()} files`,
|
||||
});
|
||||
}
|
||||
|
||||
const stats = runScopeResolution(
|
||||
{
|
||||
graph: ctx.graph,
|
||||
model,
|
||||
files,
|
||||
resolutionConfig,
|
||||
prebuiltNodeLookup: sharedNodeLookup,
|
||||
prebuiltFunctionNodeIndex: sharedFnNodeIndex,
|
||||
preExtractedParsedFiles: preExtractedByPath,
|
||||
scopeIndexStorePath: parsedFileStorePath,
|
||||
// CFG/PDG emission (#2081 M1) — opt-in; off ⇒ byte-identical graph.
|
||||
pdg: ctx.options?.pdg === true,
|
||||
pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction,
|
||||
pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction,
|
||||
pdgMaxCdgEdgesPerFunction: ctx.options?.pdgMaxCdgEdgesPerFunction,
|
||||
pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction,
|
||||
pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops,
|
||||
recordResolutionOutcome: (outcome) => {
|
||||
resolutionOutcomes.push(outcome);
|
||||
},
|
||||
onWarn: (msg) => {
|
||||
if (isSemanticModelValidatorEnabled()) {
|
||||
logger.warn(`[scope-resolution:${lang}] ${msg}`);
|
||||
}
|
||||
},
|
||||
onProgress:
|
||||
totalScopeFiles > 0
|
||||
? (subPhase: ScopeResolutionSubPhase, current, total) => {
|
||||
let langRatio: number;
|
||||
switch (subPhase) {
|
||||
case 'extracting':
|
||||
langRatio = total > 0 ? (current / total) * 0.5 : 0;
|
||||
break;
|
||||
case 'analyzing types':
|
||||
langRatio = 0.5;
|
||||
break;
|
||||
case 'resolving references':
|
||||
langRatio = 0.7;
|
||||
break;
|
||||
case 'linking symbols':
|
||||
langRatio = 0.85;
|
||||
break;
|
||||
default: {
|
||||
const _exhaustive: never = subPhase;
|
||||
langRatio = 0.85;
|
||||
}
|
||||
}
|
||||
const overallRatio = Math.min(
|
||||
1,
|
||||
(processedScopeFiles + langRatio * langFileCount) / totalScopeFiles,
|
||||
);
|
||||
const pct = SCOPE_PCT_START + Math.round(overallRatio * SCOPE_PCT_RANGE);
|
||||
ctx.onProgress({
|
||||
phase: 'scopeResolution',
|
||||
percent: pct,
|
||||
message: 'Resolving types',
|
||||
detail:
|
||||
subPhase === 'extracting'
|
||||
? `${langTag} — extracting ${current.toLocaleString()}/${total.toLocaleString()} files`
|
||||
: `${langTag} — ${subPhase}`,
|
||||
});
|
||||
}
|
||||
: undefined,
|
||||
},
|
||||
provider,
|
||||
);
|
||||
|
||||
// Release file contents and pre-extracted entries after each language
|
||||
// to reduce memory pressure. For large codebases (16K+ PHP files),
|
||||
// holding all source code simultaneously with scope trees causes OOM.
|
||||
// See: https://github.com/abhigyanpatwari/GitNexus/issues/1741
|
||||
//
|
||||
// Use `filePaths` (not `primaryFilePaths`) so that any context files
|
||||
// added by `collectScopeContextPaths` (e.g. TS/JS files pulled in for
|
||||
// Vue cross-file resolution) are also evicted and not held until GC.
|
||||
files.length = 0;
|
||||
contents.clear();
|
||||
for (const fp of filePaths) {
|
||||
preExtractedByPath.delete(fp);
|
||||
}
|
||||
// This language's ParsedFiles are now unreachable (runScopeResolution has
|
||||
// returned and the Map entries are deleted). Force a GC HERE so a heavy
|
||||
// language's ~17-20GB set (e.g. C/C++ on the Linux kernel) is reclaimed
|
||||
// BEFORE the next language's store-load — instead of leaving V8 to collect
|
||||
// it lazily under the next pass's allocation pressure (which, at a cap >=
|
||||
// RAM, degrades into swap-thrash). Collects only dead objects: the live
|
||||
// cross-file index of the next pass is untouched. The pre/post probe
|
||||
// confirms whether old-space fragmentation defeats the reclaim.
|
||||
logHeapProbe('lang-release-pre-gc', `lang=${lang}`);
|
||||
forceGc();
|
||||
logHeapProbe('lang-release-post-gc', `lang=${lang}`);
|
||||
logHeapProbe('scope-lang-end', `lang=${lang} filesProcessed=${stats.filesProcessed}`);
|
||||
|
||||
processedScopeFiles += langFileCount;
|
||||
anyRan = true;
|
||||
functionSummaries.push(...stats.functionSummaries);
|
||||
totalFiles += stats.filesProcessed;
|
||||
totalImports += stats.importsEmitted;
|
||||
totalRefs += stats.referenceEdgesEmitted;
|
||||
perLanguage.set(lang, {
|
||||
filesProcessed: stats.filesProcessed,
|
||||
importsEmitted: stats.importsEmitted,
|
||||
referenceEdgesEmitted: stats.referenceEdgesEmitted,
|
||||
});
|
||||
|
||||
if (isDev) {
|
||||
logger.info(
|
||||
`[scope-resolution:${lang}] ${stats.filesProcessed} files → ${stats.importsEmitted} IMPORTS + ${stats.referenceEdgesEmitted} reference edges (${stats.resolve.unresolved} unresolved sites, ${stats.referenceSkipped} skipped)`,
|
||||
logger.warn(
|
||||
'[scope-resolution] streaming PDG emit requested but no storage path is ' +
|
||||
'available; falling back to in-memory whole-graph emit',
|
||||
);
|
||||
}
|
||||
}
|
||||
// Cross-pass per-file dedup set for the streaming sink (#2202): one set
|
||||
// shared across every language pass so a file emitted in two passes (e.g. a
|
||||
// `.ts` module pulled into the Vue context pass) streams its PDG layer once.
|
||||
// Only created when streaming — the in-memory-graph path dedups via its Map.
|
||||
const pdgEmittedFiles = pdgEmitSink !== undefined ? new Set<string>() : undefined;
|
||||
|
||||
// Stream the PDG layer with guaranteed writer cleanup: a throw escaping the
|
||||
// per-language loop (outside run.ts's per-file try/catch — e.g. from
|
||||
// finalize/propagate/a provider hook) must still release the sink's file
|
||||
// descriptors. finalize() runs on the success path; the finally closes the
|
||||
// sink only when finalize did not (idempotent via the sink's `finalized`).
|
||||
let pdgEmitManifest: PdgEmitManifest | undefined;
|
||||
let pdgSinkSettled = false;
|
||||
try {
|
||||
for (const [lang, provider] of SCOPE_RESOLVERS) {
|
||||
// Standalone providers (COBOL, JCL) don't emit graph edges yet
|
||||
// through the scope-resolution path. This is the canonical guard:
|
||||
// runScopeResolution is never called for standalone providers, which
|
||||
// keeps cobolPhase as the sole IMPORTS edge producer. Keep this guard
|
||||
// in sync with any additional standalone providers added to
|
||||
// SCOPE_RESOLVERS.
|
||||
if (provider.languageProvider.parseStrategy === 'standalone') continue;
|
||||
|
||||
const primaryLangFiles = filesByLang.get(lang) ?? [];
|
||||
if (primaryLangFiles.length === 0) continue;
|
||||
const primaryFilePaths = primaryLangFiles.map((f) => f.path);
|
||||
|
||||
// Load per-language import-resolution config (tsconfig paths,
|
||||
// composer.json autoload, go.mod, ...). One I/O round trip per
|
||||
// workspace pass — cached implicitly by the result handed to
|
||||
// every `resolveImportTarget` call below.
|
||||
const resolutionConfig =
|
||||
provider.loadResolutionConfig !== undefined
|
||||
? await provider.loadResolutionConfig(ctx.repoPath)
|
||||
: undefined;
|
||||
|
||||
// Some languages (e.g. Vue) expand their file universe beyond the
|
||||
// primary-language files via the `collectScopeContextPaths` hook.
|
||||
// The hook receives raw source contents of the primary files so it
|
||||
// can trace import closures without a second tree-sitter parse.
|
||||
//
|
||||
// To avoid reading primary files twice (once for the hook, once for
|
||||
// the resolution pass), we read them upfront and merge with the
|
||||
// extra context paths the hook may add.
|
||||
// Stream this language's pre-built ParsedFiles in from the disk store
|
||||
// FIRST (huge-repo path). Doing it before reading source lets us skip
|
||||
// loading content for files the store already covers — for a provider
|
||||
// with no content-consuming hook that source is pure dead weight once
|
||||
// extraction is served from the store (~1.5 GB on the kernel's C pass).
|
||||
// Merged into `preExtractedByPath`; the per-language release block below
|
||||
// evicts these again before the next language, so only one language's
|
||||
// ParsedFiles are resident at a time.
|
||||
const loadStoreFor = async (paths: ReadonlySet<string>): Promise<void> => {
|
||||
if (!parsedFileStorePath) return;
|
||||
const fromDisk = await loadParsedFilesForPaths(parsedFileStorePath, paths);
|
||||
for (const [fp, pf] of fromDisk) preExtractedByPath.set(fp, pf);
|
||||
};
|
||||
|
||||
// A provider that feeds source text into a post-extract hook
|
||||
// (populateWorkspaceOwners / populateNamespaceSiblings /
|
||||
// populateRangeBindings / emitPostResolutionEdges) needs content for ALL
|
||||
// its files; one without those hooks only needs content for files the
|
||||
// store does NOT cover (fresh-extract fallback). Keep this in sync with
|
||||
// the getFileContents() call-sites in run.ts.
|
||||
const providerNeedsAllContent =
|
||||
provider.populateWorkspaceOwners !== undefined ||
|
||||
provider.populateNamespaceSiblings !== undefined ||
|
||||
provider.populateRangeBindings !== undefined ||
|
||||
provider.emitPostResolutionEdges !== undefined;
|
||||
|
||||
let scopeFilePaths: Set<string>;
|
||||
let contents: Map<string, string>;
|
||||
if (provider.collectScopeContextPaths !== undefined) {
|
||||
// Context-expanding providers (e.g. Vue) need every primary file's
|
||||
// source up front for the closure hook, so load it all.
|
||||
const entryFileContents = await readFileContents(ctx.repoPath, primaryFilePaths);
|
||||
scopeFilePaths = provider.collectScopeContextPaths({
|
||||
primaryFilePaths,
|
||||
preExtractedByPath,
|
||||
entryFileContents,
|
||||
allScannedPaths,
|
||||
resolutionConfig,
|
||||
});
|
||||
// Read only the extra context files (TS/JS etc.) not already loaded.
|
||||
const extraPaths = [...scopeFilePaths].filter((p) => !entryFileContents.has(p));
|
||||
const extraContents = await readFileContents(ctx.repoPath, extraPaths);
|
||||
contents = new Map([...entryFileContents, ...extraContents]);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
} else {
|
||||
scopeFilePaths = new Set(primaryFilePaths);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
const pathsToRead = providerNeedsAllContent
|
||||
? primaryFilePaths
|
||||
: primaryFilePaths.filter((p) => !preExtractedByPath.has(p));
|
||||
contents = await readFileContents(ctx.repoPath, pathsToRead);
|
||||
}
|
||||
const filePaths = [...scopeFilePaths];
|
||||
const files: { path: string; content: string }[] = [];
|
||||
for (const fp of filePaths) {
|
||||
const content = contents.get(fp);
|
||||
if (content !== undefined) {
|
||||
files.push({ path: fp, content });
|
||||
} else if (preExtractedByPath.has(fp)) {
|
||||
// Store covers extraction for this file and we deliberately skipped
|
||||
// reading its source; the empty string is never consumed (the
|
||||
// extract loop uses the pre-extracted ParsedFile and this provider
|
||||
// has no content hook).
|
||||
files.push({ path: fp, content: '' });
|
||||
}
|
||||
// else: uncovered AND unreadable → skip (unchanged from prior behavior).
|
||||
}
|
||||
|
||||
const langFileCount = files.length;
|
||||
logHeapProbe(
|
||||
'scope-lang-start',
|
||||
`lang=${lang} files=${langFileCount} contentsLoaded=${contents.size}`,
|
||||
);
|
||||
const langLabel = lang.charAt(0).toUpperCase() + lang.slice(1);
|
||||
currentLangIdx++;
|
||||
const langTag =
|
||||
totalScopeLangs > 1 ? `${langLabel} [${currentLangIdx}/${totalScopeLangs}]` : langLabel;
|
||||
|
||||
if (totalScopeFiles > 0) {
|
||||
const pct =
|
||||
SCOPE_PCT_START + Math.round((processedScopeFiles / totalScopeFiles) * SCOPE_PCT_RANGE);
|
||||
ctx.onProgress({
|
||||
phase: 'scopeResolution',
|
||||
percent: pct,
|
||||
message: 'Resolving types',
|
||||
detail: `${langTag}, ${langFileCount.toLocaleString()} files`,
|
||||
});
|
||||
}
|
||||
|
||||
const stats = runScopeResolution(
|
||||
{
|
||||
graph: ctx.graph,
|
||||
model,
|
||||
files,
|
||||
resolutionConfig,
|
||||
prebuiltNodeLookup: sharedNodeLookup,
|
||||
prebuiltFunctionNodeIndex: sharedFnNodeIndex,
|
||||
preExtractedParsedFiles: preExtractedByPath,
|
||||
scopeIndexStorePath: parsedFileStorePath,
|
||||
// CFG/PDG emission (#2081 M1) — opt-in; off ⇒ byte-identical graph.
|
||||
pdg: ctx.options?.pdg === true,
|
||||
pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction,
|
||||
pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction,
|
||||
pdgMaxCdgEdgesPerFunction: ctx.options?.pdgMaxCdgEdgesPerFunction,
|
||||
pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction,
|
||||
pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops,
|
||||
// Streaming PDG-emit sink (#2202) — undefined ⇒ emit to the in-memory graph.
|
||||
pdgEmitSink,
|
||||
// Cross-pass per-file dedup set (#2202) — undefined when not streaming.
|
||||
pdgEmittedFiles,
|
||||
recordResolutionOutcome: (outcome) => {
|
||||
resolutionOutcomes.push(outcome);
|
||||
},
|
||||
onWarn: (msg) => {
|
||||
if (isSemanticModelValidatorEnabled()) {
|
||||
logger.warn(`[scope-resolution:${lang}] ${msg}`);
|
||||
}
|
||||
},
|
||||
onProgress:
|
||||
totalScopeFiles > 0
|
||||
? (subPhase: ScopeResolutionSubPhase, current, total) => {
|
||||
let langRatio: number;
|
||||
switch (subPhase) {
|
||||
case 'extracting':
|
||||
langRatio = total > 0 ? (current / total) * 0.5 : 0;
|
||||
break;
|
||||
case 'analyzing types':
|
||||
langRatio = 0.5;
|
||||
break;
|
||||
case 'resolving references':
|
||||
langRatio = 0.7;
|
||||
break;
|
||||
case 'linking symbols':
|
||||
langRatio = 0.85;
|
||||
break;
|
||||
default: {
|
||||
const _exhaustive: never = subPhase;
|
||||
langRatio = 0.85;
|
||||
}
|
||||
}
|
||||
const overallRatio = Math.min(
|
||||
1,
|
||||
(processedScopeFiles + langRatio * langFileCount) / totalScopeFiles,
|
||||
);
|
||||
const pct = SCOPE_PCT_START + Math.round(overallRatio * SCOPE_PCT_RANGE);
|
||||
ctx.onProgress({
|
||||
phase: 'scopeResolution',
|
||||
percent: pct,
|
||||
message: 'Resolving types',
|
||||
detail:
|
||||
subPhase === 'extracting'
|
||||
? `${langTag} — extracting ${current.toLocaleString()}/${total.toLocaleString()} files`
|
||||
: `${langTag} — ${subPhase}`,
|
||||
});
|
||||
}
|
||||
: undefined,
|
||||
},
|
||||
provider,
|
||||
);
|
||||
|
||||
// Release file contents and pre-extracted entries after each language
|
||||
// to reduce memory pressure. For large codebases (16K+ PHP files),
|
||||
// holding all source code simultaneously with scope trees causes OOM.
|
||||
// See: https://github.com/abhigyanpatwari/GitNexus/issues/1741
|
||||
//
|
||||
// Use `filePaths` (not `primaryFilePaths`) so that any context files
|
||||
// added by `collectScopeContextPaths` (e.g. TS/JS files pulled in for
|
||||
// Vue cross-file resolution) are also evicted and not held until GC.
|
||||
files.length = 0;
|
||||
contents.clear();
|
||||
for (const fp of filePaths) {
|
||||
preExtractedByPath.delete(fp);
|
||||
}
|
||||
// This language's ParsedFiles are now unreachable (runScopeResolution has
|
||||
// returned and the Map entries are deleted). Force a GC HERE so a heavy
|
||||
// language's ~17-20GB set (e.g. C/C++ on the Linux kernel) is reclaimed
|
||||
// BEFORE the next language's store-load — instead of leaving V8 to collect
|
||||
// it lazily under the next pass's allocation pressure (which, at a cap >=
|
||||
// RAM, degrades into swap-thrash). Collects only dead objects: the live
|
||||
// cross-file index of the next pass is untouched. The pre/post probe
|
||||
// confirms whether old-space fragmentation defeats the reclaim.
|
||||
logHeapProbe('lang-release-pre-gc', `lang=${lang}`);
|
||||
forceGc();
|
||||
logHeapProbe('lang-release-post-gc', `lang=${lang}`);
|
||||
logHeapProbe('scope-lang-end', `lang=${lang} filesProcessed=${stats.filesProcessed}`);
|
||||
|
||||
processedScopeFiles += langFileCount;
|
||||
anyRan = true;
|
||||
functionSummaries.push(...stats.functionSummaries);
|
||||
totalFiles += stats.filesProcessed;
|
||||
totalImports += stats.importsEmitted;
|
||||
totalRefs += stats.referenceEdgesEmitted;
|
||||
perLanguage.set(lang, {
|
||||
filesProcessed: stats.filesProcessed,
|
||||
importsEmitted: stats.importsEmitted,
|
||||
referenceEdgesEmitted: stats.referenceEdgesEmitted,
|
||||
});
|
||||
|
||||
if (isDev) {
|
||||
logger.info(
|
||||
`[scope-resolution:${lang}] ${stats.filesProcessed} files → ${stats.importsEmitted} IMPORTS + ${stats.referenceEdgesEmitted} reference edges (${stats.resolve.unresolved} unresolved sites, ${stats.referenceSkipped} skipped)`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Finalize the streaming PDG sink (#2202) once after the last language:
|
||||
// flush + close its CSV writers and capture the COPY manifest. forceGc at
|
||||
// the boundary reclaims transient write buffers (mirrors the per-language
|
||||
// release below).
|
||||
pdgEmitManifest = pdgEmitSink?.finalize();
|
||||
pdgSinkSettled = true;
|
||||
if (pdgEmitSink !== undefined) forceGc();
|
||||
} finally {
|
||||
// Release fds if a throw skipped finalize (idempotent with finalize()).
|
||||
if (pdgEmitSink !== undefined && !pdgSinkSettled) pdgEmitSink.close();
|
||||
}
|
||||
|
||||
if (totalScopeFiles > 0 && anyRan) {
|
||||
ctx.onProgress({
|
||||
|
|
@ -494,7 +559,10 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
}
|
||||
}
|
||||
|
||||
if (!anyRan) return NOOP_OUTPUT;
|
||||
// Even when no language ran, surface a finalized manifest (its CSVs are on
|
||||
// disk) so loadGraphToLbug COPYs them rather than orphaning them — empty in
|
||||
// the no-files case, harmless.
|
||||
if (!anyRan) return pdgEmitManifest ? { ...NOOP_OUTPUT, pdgEmitManifest } : NOOP_OUTPUT;
|
||||
|
||||
return {
|
||||
ran: true,
|
||||
|
|
@ -504,6 +572,7 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
resolutionOutcomes,
|
||||
perLanguage,
|
||||
functionSummaries,
|
||||
pdgEmitManifest,
|
||||
};
|
||||
},
|
||||
};
|
||||
|
|
|
|||
|
|
@ -302,6 +302,30 @@ interface RunScopeResolutionInput {
|
|||
* `reason`; consumed by the U4 taint emit step). `undefined` ⇒
|
||||
* `DEFAULT_PDG_MAX_TAINT_HOPS` (32); `0` ⇒ no cap. */
|
||||
readonly pdgMaxTaintHops?: number;
|
||||
/**
|
||||
* Streaming PDG-emit sink (#2202). When present (streaming on, full rebuild),
|
||||
* the `--pdg` emit routes BasicBlock nodes + intra-file PDG edges to THIS
|
||||
* graph-shaped target instead of the in-memory `graph`, so the bulky PDG
|
||||
* layer never accumulates in memory (peak RSS O(chunk)). Typed as a plain
|
||||
* `KnowledgeGraph` so this module stays decoupled from the persistence layer;
|
||||
* the caller (the scope-resolution phase) owns its lifecycle and finalizes it
|
||||
* after the last language. Absent ⇒ the emit writes to `graph` as before
|
||||
* (byte-identical default).
|
||||
*/
|
||||
readonly pdgEmitSink?: KnowledgeGraph;
|
||||
/**
|
||||
* Cross-pass per-file dedup set for streaming PDG emit (#2202). Shared across
|
||||
* every language pass (owned by the scope-resolution phase). A file imported
|
||||
* by more than one language (e.g. a `.ts` module pulled into the Vue context
|
||||
* pass) is PDG-emitted in each pass over the same `cfgSideChannel`, producing
|
||||
* identical ids; the in-memory graph dedups that by id, but the streaming sink
|
||||
* is dedup-free (to stay O(write buffer), not O(total ids)). So when present
|
||||
* (streaming on), the emit loop skips a file whose PDG already streamed and
|
||||
* records the rest — keeping the streamed set byte-identical to the
|
||||
* Map-deduped whole-graph emit, for any language-pass order. Absent ⇒ no skip
|
||||
* (the graph Map dedups), so the default path is unchanged.
|
||||
*/
|
||||
readonly pdgEmittedFiles?: Set<string>;
|
||||
/**
|
||||
* Optional graph-node lookup built ONCE by the caller and shared across
|
||||
* every language pass. `buildGraphNodeLookup` scans the whole graph and is
|
||||
|
|
@ -769,6 +793,11 @@ export function runScopeResolution(
|
|||
// can bracket it. Printed as the PROF `taint=` segment.
|
||||
let taintMs = 0;
|
||||
if (input.pdg === true) {
|
||||
// Streaming target (#2202): when a sink is provided, BasicBlock nodes +
|
||||
// intra-file PDG edges are routed to CSV-on-disk through it instead of
|
||||
// accumulating in `graph`. The function-node index below is still built
|
||||
// from the real `graph` (Function/Method nodes live there, never the sink).
|
||||
const pdgTarget: KnowledgeGraph = input.pdgEmitSink ?? graph;
|
||||
let cfgBlocks = 0;
|
||||
let cfgEdges = 0;
|
||||
let cfgDroppedEdges = 0;
|
||||
|
|
@ -835,6 +864,15 @@ export function runScopeResolution(
|
|||
// shard that slipped the version gate) must skip emission, not throw a
|
||||
// TypeError mid-graph-build and abort scope-resolution for the language.
|
||||
if (!Array.isArray(cfgs) || cfgs.length === 0) continue;
|
||||
// Cross-pass per-file dedup (#2202): when streaming, a file whose PDG
|
||||
// already streamed in a prior language pass (e.g. a `.ts` module pulled
|
||||
// into the Vue context pass) would re-emit identical ids from the same
|
||||
// cfgSideChannel — the dedup-free streaming sink would double the rows.
|
||||
// Skip it here; the in-memory-graph path needs no skip (its Map dedups).
|
||||
if (input.pdgEmittedFiles !== undefined) {
|
||||
if (input.pdgEmittedFiles.has(pf.filePath)) continue;
|
||||
input.pdgEmittedFiles.add(pf.filePath);
|
||||
}
|
||||
try {
|
||||
// Per-element emit-safety filter (mirrors the parsedfile-store
|
||||
// reviver's POLICY: valid elements in a mixed array still emit; junk
|
||||
|
|
@ -854,7 +892,7 @@ export function runScopeResolution(
|
|||
}
|
||||
if (wellFormed.length === 0) continue;
|
||||
const emitted = emitFileCfgs(
|
||||
graph,
|
||||
pdgTarget,
|
||||
wellFormed,
|
||||
input.pdgMaxEdgesPerFunction ?? DEFAULT_MAX_CFG_EDGES_PER_FUNCTION,
|
||||
// Log cap-overflow drops UNCONDITIONALLY (not via input.onWarn, which is
|
||||
|
|
@ -873,7 +911,7 @@ export function runScopeResolution(
|
|||
// PROF-gated like every other checkpoint here (zero cost when off).
|
||||
const t0 = PROF ? performance.now() : 0;
|
||||
const rd = emitFileReachingDefs(
|
||||
graph,
|
||||
pdgTarget,
|
||||
wellFormed,
|
||||
input.pdgMaxReachingDefEdgesPerFunction ??
|
||||
DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION,
|
||||
|
|
@ -892,7 +930,7 @@ export function runScopeResolution(
|
|||
// persisted and its time folds into the `pdg=` PROF segment next to RD.
|
||||
const tCdg = PROF ? performance.now() : 0;
|
||||
const cdg = emitFileCdg(
|
||||
graph,
|
||||
pdgTarget,
|
||||
wellFormed,
|
||||
input.pdgMaxCdgEdgesPerFunction ?? DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION,
|
||||
(message) => logger.warn(message), // unconditional — R6, no silent truncation
|
||||
|
|
@ -909,7 +947,7 @@ export function runScopeResolution(
|
|||
if (taintSpec !== undefined) {
|
||||
const t1 = PROF ? performance.now() : 0;
|
||||
const taint = emitFileTaint(
|
||||
graph,
|
||||
pdgTarget,
|
||||
wellFormed,
|
||||
pf.parsedImports,
|
||||
taintSpec,
|
||||
|
|
|
|||
|
|
@ -28,6 +28,18 @@ export const parseTruthyEnv = (raw: string | undefined): boolean => {
|
|||
return value === '1' || value === 'true' || value === 'yes';
|
||||
};
|
||||
|
||||
/**
|
||||
* Parse a positive-integer env-var value. Returns the integer when `raw` is a
|
||||
* finite integer `> 0`; otherwise (`undefined`, empty, non-numeric, `0`, or
|
||||
* negative) returns `undefined` so the caller falls back to its default. Used
|
||||
* for numeric tuning knobs like `GITNEXUS_PDG_EMIT_CHUNK_SIZE` (#2202).
|
||||
*/
|
||||
export const parsePositiveIntEnv = (raw: string | undefined): number | undefined => {
|
||||
if (raw === undefined) return undefined;
|
||||
const n = Number(raw.trim());
|
||||
return Number.isInteger(n) && n > 0 ? n : undefined;
|
||||
};
|
||||
|
||||
/**
|
||||
* Whether scope-resolution dev validators (e.g. `validateBindingsImmutability`)
|
||||
* should run AND emit warnings. Off by default in CLI runs to avoid silent
|
||||
|
|
|
|||
|
|
@ -17,7 +17,8 @@ import { createWriteStream, WriteStream } from 'fs';
|
|||
import path from 'path';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
import { KnowledgeGraph } from '../graph/types.js';
|
||||
import { NodeTableName } from './schema.js';
|
||||
import { NodeTableName, NODE_TABLES } from './schema.js';
|
||||
import { RelPairRouter } from './rel-pair-routing.js';
|
||||
import { parseTruthyEnv } from '../ingestion/utils/env.js';
|
||||
|
||||
/**
|
||||
|
|
@ -182,13 +183,20 @@ class BufferedCSVWriter {
|
|||
this.buffer.push(header);
|
||||
}
|
||||
|
||||
addRow(row: string) {
|
||||
/**
|
||||
* Buffer a row. Returns a promise ONLY when the buffer crossed FLUSH_EVERY
|
||||
* and a disk write was issued; otherwise returns `undefined` so the caller
|
||||
* can skip awaiting (#2203 U3) — avoiding a microtask tick on every buffered
|
||||
* row (millions at scale). The flush promise still resolves on drain, so
|
||||
* backpressure is preserved on the rows that actually write.
|
||||
*/
|
||||
addRow(row: string): Promise<void> | undefined {
|
||||
this.buffer.push(row);
|
||||
this.rows++;
|
||||
if (this.buffer.length >= FLUSH_EVERY) {
|
||||
return this.flush();
|
||||
}
|
||||
return Promise.resolve();
|
||||
return undefined;
|
||||
}
|
||||
|
||||
flush(): Promise<void> {
|
||||
|
|
@ -223,10 +231,51 @@ class BufferedCSVWriter {
|
|||
// STREAMING CSV GENERATION — SINGLE PASS
|
||||
// ============================================================================
|
||||
|
||||
/** Canonical relationship CSV header — shared by the emit pass and the
|
||||
* `splitRelCsvByLabelPair` differential oracle. */
|
||||
export const REL_CSV_HEADER = 'from,to,type,confidence,reason,step';
|
||||
|
||||
/** Build the escaped CSV row (no trailing newline) for one relationship.
|
||||
* Single source of the relationship row bytes — used by the emit pass and by
|
||||
* the byte-identity differential test that feeds the legacy split oracle. */
|
||||
export const buildRelRow = (rel: GraphRelationship): string =>
|
||||
[
|
||||
escapeCSVField(rel.sourceId),
|
||||
escapeCSVField(rel.targetId),
|
||||
escapeCSVField(rel.type),
|
||||
escapeCSVNumber(rel.confidence, 1.0),
|
||||
escapeCSVField(rel.reason),
|
||||
escapeCSVNumber(rel.step, 0),
|
||||
].join(',');
|
||||
|
||||
/** Canonical BasicBlock node CSV header — taint/PDG substrate (issue #2080).
|
||||
* No `name` column; blocks are identified by id + source span. Shared by the
|
||||
* whole-graph emit pass and the streaming PDG emit sink (issue #2202) so the
|
||||
* two paths produce byte-identical BasicBlock rows by construction. */
|
||||
export const BASICBLOCK_CSV_HEADER = 'id,filePath,startLine,endLine,text';
|
||||
|
||||
/** Build the escaped CSV row (no trailing newline) for one BasicBlock node.
|
||||
* Single source of the BasicBlock row bytes — used by `streamAllCSVsToDisk`
|
||||
* and by the streaming `PdgEmitSink` (issue #2202). */
|
||||
export const buildBasicBlockRow = (node: GraphNode): string =>
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVField(node.properties.text || ''),
|
||||
].join(',');
|
||||
|
||||
export interface StreamedCSVResult {
|
||||
nodeFiles: Map<NodeTableName, { csvPath: string; rows: number }>;
|
||||
relCsvPath: string;
|
||||
relRows: number;
|
||||
/** pairKey (`From|To`) → per-FROM→TO-label-pair CSV file. */
|
||||
relsByPair: Map<string, { csvPath: string; rows: number }>;
|
||||
/** Header line shared by every per-pair file. */
|
||||
relHeader: string;
|
||||
/** Edges skipped because an endpoint label is not a valid node table. */
|
||||
skippedRels: number;
|
||||
/** Edges routed to a per-pair file. */
|
||||
totalValidRels: number;
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -253,255 +302,185 @@ export const streamAllCSVsToDisk = async (
|
|||
const prevMax = process.getMaxListeners();
|
||||
process.setMaxListeners(prevMax + 40);
|
||||
|
||||
const contentCache = new FileContentCache(repoPath);
|
||||
// try/finally so the listener bump is ALWAYS restored — including the
|
||||
// rel-routing throw path (#2203 U2) and any node-writer finish() rejection,
|
||||
// not just the success path (avoids leaking +40 listeners across failed runs
|
||||
// in long-lived hosts / the test suite).
|
||||
try {
|
||||
const contentCache = new FileContentCache(repoPath);
|
||||
|
||||
// Create writers for every node type up-front
|
||||
const fileWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'file.csv'),
|
||||
'id,name,filePath,content',
|
||||
);
|
||||
const folderWriter = new BufferedCSVWriter(path.join(csvDir, 'folder.csv'), 'id,name,filePath');
|
||||
const codeElementHeader = 'id,name,filePath,startLine,endLine,isExported,content,description';
|
||||
const functionWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'function.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const classWriter = new BufferedCSVWriter(path.join(csvDir, 'class.csv'), codeElementHeader);
|
||||
const interfaceWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'interface.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const methodHeader =
|
||||
'id,name,filePath,startLine,endLine,isExported,content,description,parameterCount,returnType';
|
||||
const methodWriter = new BufferedCSVWriter(path.join(csvDir, 'method.csv'), methodHeader);
|
||||
const codeElemWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'codeelement.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const communityWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'community.csv'),
|
||||
'id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount',
|
||||
);
|
||||
const processWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'process.csv'),
|
||||
'id,label,heuristicLabel,processType,stepCount,communities,entryPointId,terminalId',
|
||||
);
|
||||
|
||||
// Section nodes have an extra 'level' column
|
||||
const sectionWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'section.csv'),
|
||||
'id,name,filePath,startLine,endLine,level,content,description',
|
||||
);
|
||||
|
||||
// Route nodes for API endpoint mapping
|
||||
const routeWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'route.csv'),
|
||||
'id,name,filePath,responseKeys,errorKeys,middleware',
|
||||
);
|
||||
|
||||
// Tool nodes for MCP tool definitions
|
||||
const toolWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'tool.csv'),
|
||||
'id,name,filePath,description',
|
||||
);
|
||||
|
||||
// BasicBlock nodes — taint/PDG substrate (issue #2080). No `name` column;
|
||||
// blocks are identified by id + source span. Emitted by no phase yet.
|
||||
const basicBlockWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'basicblock.csv'),
|
||||
'id,filePath,startLine,endLine,text',
|
||||
);
|
||||
|
||||
// Multi-language node types share the same CSV shape (no isExported column)
|
||||
const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description';
|
||||
const MULTI_LANG_TYPES = [
|
||||
'Struct',
|
||||
'Enum',
|
||||
'Macro',
|
||||
'Typedef',
|
||||
'Union',
|
||||
'Namespace',
|
||||
'Trait',
|
||||
'Impl',
|
||||
'TypeAlias',
|
||||
'Const',
|
||||
'Static',
|
||||
'Variable',
|
||||
'Property',
|
||||
'Record',
|
||||
'Delegate',
|
||||
'Annotation',
|
||||
'Constructor',
|
||||
'Template',
|
||||
'Module',
|
||||
] as const;
|
||||
const propertyHeader = 'id,name,filePath,startLine,endLine,content,description,declaredType';
|
||||
const multiLangWriters = new Map<string, BufferedCSVWriter>();
|
||||
for (const t of MULTI_LANG_TYPES) {
|
||||
multiLangWriters.set(
|
||||
t,
|
||||
new BufferedCSVWriter(
|
||||
path.join(csvDir, `${t.toLowerCase()}.csv`),
|
||||
t === 'Property' ? propertyHeader : multiLangHeader,
|
||||
),
|
||||
// Create writers for every node type up-front
|
||||
const fileWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'file.csv'),
|
||||
'id,name,filePath,content',
|
||||
);
|
||||
const folderWriter = new BufferedCSVWriter(path.join(csvDir, 'folder.csv'), 'id,name,filePath');
|
||||
const codeElementHeader = 'id,name,filePath,startLine,endLine,isExported,content,description';
|
||||
const functionWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'function.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const classWriter = new BufferedCSVWriter(path.join(csvDir, 'class.csv'), codeElementHeader);
|
||||
const interfaceWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'interface.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const methodHeader =
|
||||
'id,name,filePath,startLine,endLine,isExported,content,description,parameterCount,returnType';
|
||||
const methodWriter = new BufferedCSVWriter(path.join(csvDir, 'method.csv'), methodHeader);
|
||||
const codeElemWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'codeelement.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const communityWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'community.csv'),
|
||||
'id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount',
|
||||
);
|
||||
const processWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'process.csv'),
|
||||
'id,label,heuristicLabel,processType,stepCount,communities,entryPointId,terminalId',
|
||||
);
|
||||
}
|
||||
|
||||
const codeWriterMap: Record<string, BufferedCSVWriter> = {
|
||||
Function: functionWriter,
|
||||
Class: classWriter,
|
||||
Interface: interfaceWriter,
|
||||
CodeElement: codeElemWriter,
|
||||
};
|
||||
// Section nodes have an extra 'level' column
|
||||
const sectionWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'section.csv'),
|
||||
'id,name,filePath,startLine,endLine,level,content,description',
|
||||
);
|
||||
|
||||
// Deduplicate all node types — the pipeline can produce duplicate IDs across
|
||||
// all symbol types (Class, Method, Function, etc.), not just File nodes.
|
||||
// A single Set covering every label prevents PK violations on COPY.
|
||||
const seenNodeIds = new Set<string>();
|
||||
// Route nodes for API endpoint mapping
|
||||
const routeWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'route.csv'),
|
||||
'id,name,filePath,responseKeys,errorKeys,middleware',
|
||||
);
|
||||
|
||||
// --- SINGLE PASS over all nodes ---
|
||||
for (const node of orderedNodes(graph, sortOutput)) {
|
||||
if (seenNodeIds.has(node.id)) continue;
|
||||
seenNodeIds.add(node.id);
|
||||
// Tool nodes for MCP tool definitions
|
||||
const toolWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'tool.csv'),
|
||||
'id,name,filePath,description',
|
||||
);
|
||||
|
||||
switch (node.label) {
|
||||
case 'File': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await fileWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(content),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Folder':
|
||||
await folderWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
case 'Community': {
|
||||
const keywords = node.properties.keywords || [];
|
||||
const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/\\/g, '\\\\').replace(/'/g, "''").replace(/,/g, '\\,')}'`).join(',')}]`;
|
||||
await communityWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
keywordsStr,
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
escapeCSVField(node.properties.enrichedBy || 'heuristic'),
|
||||
escapeCSVNumber(node.properties.cohesion, 0),
|
||||
escapeCSVNumber(node.properties.symbolCount, 0),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Process': {
|
||||
const communities = node.properties.communities || [];
|
||||
const communitiesStr = `[${communities.map((c: string) => `'${c.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
await processWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
escapeCSVField(node.properties.processType || ''),
|
||||
escapeCSVNumber(node.properties.stepCount, 0),
|
||||
escapeCSVField(communitiesStr),
|
||||
escapeCSVField(node.properties.entryPointId || ''),
|
||||
escapeCSVField(node.properties.terminalId || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Method': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await methodWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
node.properties.isExported ? 'true' : 'false',
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
escapeCSVNumber(node.properties.parameterCount, 0),
|
||||
escapeCSVField(node.properties.returnType || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Section': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await sectionWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVNumber(node.properties.level, 1),
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Route': {
|
||||
const responseKeys = node.properties.responseKeys || [];
|
||||
// LadybugDB array literal inside a quoted CSV field: escapeCSVField wraps in "..."
|
||||
// and the array uses single-quoted elements
|
||||
const keysStr = `[${responseKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
const errorKeys = node.properties.errorKeys || [];
|
||||
const errorKeysStr = `[${errorKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
const middleware = node.properties.middleware || [];
|
||||
const middlewareStr = `[${middleware.map((m: string) => `'${m.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
await routeWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(keysStr),
|
||||
escapeCSVField(errorKeysStr),
|
||||
escapeCSVField(middlewareStr),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Tool':
|
||||
await toolWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
case 'BasicBlock':
|
||||
await basicBlockWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVField(node.properties.text || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
default: {
|
||||
// Code element nodes (Function, Class, Interface, CodeElement)
|
||||
const writer = codeWriterMap[node.label];
|
||||
if (writer) {
|
||||
// BasicBlock nodes — taint/PDG substrate (issue #2080). No `name` column;
|
||||
// blocks are identified by id + source span. Emitted by no phase yet.
|
||||
const basicBlockWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'basicblock.csv'),
|
||||
BASICBLOCK_CSV_HEADER,
|
||||
);
|
||||
|
||||
// Multi-language node types share the same CSV shape (no isExported column)
|
||||
const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description';
|
||||
const MULTI_LANG_TYPES = [
|
||||
'Struct',
|
||||
'Enum',
|
||||
'Macro',
|
||||
'Typedef',
|
||||
'Union',
|
||||
'Namespace',
|
||||
'Trait',
|
||||
'Impl',
|
||||
'TypeAlias',
|
||||
'Const',
|
||||
'Static',
|
||||
'Variable',
|
||||
'Property',
|
||||
'Record',
|
||||
'Delegate',
|
||||
'Annotation',
|
||||
'Constructor',
|
||||
'Template',
|
||||
'Module',
|
||||
] as const;
|
||||
const propertyHeader = 'id,name,filePath,startLine,endLine,content,description,declaredType';
|
||||
const multiLangWriters = new Map<string, BufferedCSVWriter>();
|
||||
for (const t of MULTI_LANG_TYPES) {
|
||||
multiLangWriters.set(
|
||||
t,
|
||||
new BufferedCSVWriter(
|
||||
path.join(csvDir, `${t.toLowerCase()}.csv`),
|
||||
t === 'Property' ? propertyHeader : multiLangHeader,
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
const codeWriterMap: Record<string, BufferedCSVWriter> = {
|
||||
Function: functionWriter,
|
||||
Class: classWriter,
|
||||
Interface: interfaceWriter,
|
||||
CodeElement: codeElemWriter,
|
||||
};
|
||||
|
||||
// Deduplicate all node types — the pipeline can produce duplicate IDs across
|
||||
// all symbol types (Class, Method, Function, etc.), not just File nodes.
|
||||
// A single Set covering every label prevents PK violations on COPY.
|
||||
const seenNodeIds = new Set<string>();
|
||||
|
||||
// --- SINGLE PASS over all nodes ---
|
||||
for (const node of orderedNodes(graph, sortOutput)) {
|
||||
if (seenNodeIds.has(node.id)) continue;
|
||||
seenNodeIds.add(node.id);
|
||||
|
||||
// addRow returns a promise only when it flushes; awaiting it once after the
|
||||
// switch (instead of `await`-ing every addRow) skips a per-row microtask
|
||||
// tick on the ~FLUSH_EVERY-1 buffered rows between flushes (#2203 U3).
|
||||
let pending: Promise<void> | undefined;
|
||||
switch (node.label) {
|
||||
case 'File': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await writer.addRow(
|
||||
pending = fileWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(content),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Folder':
|
||||
pending = folderWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
case 'Community': {
|
||||
const keywords = node.properties.keywords || [];
|
||||
const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/\\/g, '\\\\').replace(/'/g, "''").replace(/,/g, '\\,')}'`).join(',')}]`;
|
||||
pending = communityWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
keywordsStr,
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
escapeCSVField(node.properties.enrichedBy || 'heuristic'),
|
||||
escapeCSVNumber(node.properties.cohesion, 0),
|
||||
escapeCSVNumber(node.properties.symbolCount, 0),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Process': {
|
||||
const communities = node.properties.communities || [];
|
||||
const communitiesStr = `[${communities.map((c: string) => `'${c.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
pending = processWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
escapeCSVField(node.properties.processType || ''),
|
||||
escapeCSVNumber(node.properties.stepCount, 0),
|
||||
escapeCSVField(communitiesStr),
|
||||
escapeCSVField(node.properties.entryPointId || ''),
|
||||
escapeCSVField(node.properties.terminalId || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Method': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
pending = methodWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
|
|
@ -511,101 +490,191 @@ export const streamAllCSVsToDisk = async (
|
|||
node.properties.isExported ? 'true' : 'false',
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
escapeCSVNumber(node.properties.parameterCount, 0),
|
||||
escapeCSVField(node.properties.returnType || ''),
|
||||
].join(','),
|
||||
);
|
||||
} else {
|
||||
// Multi-language node types (Struct, Impl, Trait, Macro, etc.)
|
||||
const mlWriter = multiLangWriters.get(node.label);
|
||||
if (mlWriter) {
|
||||
break;
|
||||
}
|
||||
case 'Section': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
pending = sectionWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVNumber(node.properties.level, 1),
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Route': {
|
||||
const responseKeys = node.properties.responseKeys || [];
|
||||
// LadybugDB array literal inside a quoted CSV field: escapeCSVField wraps in "..."
|
||||
// and the array uses single-quoted elements
|
||||
const keysStr = `[${responseKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
const errorKeys = node.properties.errorKeys || [];
|
||||
const errorKeysStr = `[${errorKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
const middleware = node.properties.middleware || [];
|
||||
const middlewareStr = `[${middleware.map((m: string) => `'${m.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
pending = routeWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(keysStr),
|
||||
escapeCSVField(errorKeysStr),
|
||||
escapeCSVField(middlewareStr),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Tool':
|
||||
pending = toolWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
case 'BasicBlock':
|
||||
pending = basicBlockWriter.addRow(buildBasicBlockRow(node));
|
||||
break;
|
||||
default: {
|
||||
// Code element nodes (Function, Class, Interface, CodeElement)
|
||||
const writer = codeWriterMap[node.label];
|
||||
if (writer) {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await mlWriter.addRow(
|
||||
pending = writer.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
node.properties.isExported ? 'true' : 'false',
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
...(node.label === 'Property'
|
||||
? [escapeCSVField(node.properties.declaredType || '')]
|
||||
: []),
|
||||
].join(','),
|
||||
);
|
||||
} else {
|
||||
// Multi-language node types (Struct, Impl, Trait, Macro, etc.)
|
||||
const mlWriter = multiLangWriters.get(node.label);
|
||||
if (mlWriter) {
|
||||
const content = await extractContent(node, contentCache);
|
||||
pending = mlWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
...(node.label === 'Property'
|
||||
? [escapeCSVField(node.properties.declaredType || '')]
|
||||
: []),
|
||||
].join(','),
|
||||
);
|
||||
} else {
|
||||
// Unknown label: not in codeWriterMap or multiLangWriters, so there
|
||||
// is no CSV table for it and it is intentionally NOT persisted —
|
||||
// `pending` stays undefined, so the loop awaits nothing. Made
|
||||
// explicit so a future node type isn't silently dropped here: wire
|
||||
// it into one of the writer maps above (or this branch).
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (pending) await pending;
|
||||
}
|
||||
|
||||
// Finish all node writers
|
||||
const allWriters = [
|
||||
fileWriter,
|
||||
folderWriter,
|
||||
functionWriter,
|
||||
classWriter,
|
||||
interfaceWriter,
|
||||
methodWriter,
|
||||
codeElemWriter,
|
||||
communityWriter,
|
||||
processWriter,
|
||||
sectionWriter,
|
||||
routeWriter,
|
||||
toolWriter,
|
||||
basicBlockWriter,
|
||||
...multiLangWriters.values(),
|
||||
];
|
||||
await Promise.all(allWriters.map((w) => w.finish()));
|
||||
|
||||
// --- Stream relationships directly to per-FROM→TO-label-pair files ---
|
||||
// (#2203 U2) Route every edge to its pair file in this single pass. The old
|
||||
// monolithic relations.csv — and its line-by-line re-read + per-edge regex
|
||||
// re-split in loadGraphToLbug — are gone, so the ~1M-edge set is written and
|
||||
// read once instead of twice. The router applies the SAME label-derivation +
|
||||
// validTables filter as the legacy splitRelCsvByLabelPair, so the per-pair
|
||||
// files are byte-identical (asserted by the differential test).
|
||||
const relRouter = new RelPairRouter(csvDir, REL_CSV_HEADER, new Set<string>(NODE_TABLES));
|
||||
try {
|
||||
for (const rel of orderedRelationships(graph, sortOutput)) {
|
||||
const pending = relRouter.route(rel.sourceId, rel.targetId, buildRelRow(rel));
|
||||
if (pending) await pending;
|
||||
}
|
||||
await relRouter.close();
|
||||
} catch (err) {
|
||||
relRouter.destroy();
|
||||
// Rethrow the real stream error (EMFILE / disk-full) rather than the generic
|
||||
// AbortError a pending drain-await rejects with — mirrors the retained
|
||||
// splitRelCsvByLabelPair's `throw streamError ?? err`.
|
||||
throw relRouter.lastError ?? err;
|
||||
}
|
||||
|
||||
// Build result map — only include tables that have rows
|
||||
const nodeFiles = new Map<NodeTableName, { csvPath: string; rows: number }>();
|
||||
const tableMap: [NodeTableName, BufferedCSVWriter][] = [
|
||||
['File', fileWriter],
|
||||
['Folder', folderWriter],
|
||||
['Function', functionWriter],
|
||||
['Class', classWriter],
|
||||
['Interface', interfaceWriter],
|
||||
['Method', methodWriter],
|
||||
['CodeElement', codeElemWriter],
|
||||
['Community', communityWriter],
|
||||
['Process', processWriter],
|
||||
['Section' as NodeTableName, sectionWriter],
|
||||
['Route' as NodeTableName, routeWriter],
|
||||
['Tool' as NodeTableName, toolWriter],
|
||||
['BasicBlock' as NodeTableName, basicBlockWriter],
|
||||
...Array.from(multiLangWriters.entries()).map(
|
||||
([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter],
|
||||
),
|
||||
];
|
||||
for (const [name, writer] of tableMap) {
|
||||
if (writer.rows > 0) {
|
||||
nodeFiles.set(name, {
|
||||
csvPath: path.join(csvDir, `${name.toLowerCase()}.csv`),
|
||||
rows: writer.rows,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
nodeFiles,
|
||||
relsByPair: relRouter.byPair,
|
||||
relHeader: REL_CSV_HEADER,
|
||||
skippedRels: relRouter.skipped,
|
||||
totalValidRels: relRouter.total,
|
||||
};
|
||||
} finally {
|
||||
// Restore original process listener limit on every path (success or throw).
|
||||
process.setMaxListeners(prevMax);
|
||||
}
|
||||
|
||||
// Finish all node writers
|
||||
const allWriters = [
|
||||
fileWriter,
|
||||
folderWriter,
|
||||
functionWriter,
|
||||
classWriter,
|
||||
interfaceWriter,
|
||||
methodWriter,
|
||||
codeElemWriter,
|
||||
communityWriter,
|
||||
processWriter,
|
||||
sectionWriter,
|
||||
routeWriter,
|
||||
toolWriter,
|
||||
basicBlockWriter,
|
||||
...multiLangWriters.values(),
|
||||
];
|
||||
await Promise.all(allWriters.map((w) => w.finish()));
|
||||
|
||||
// --- Stream relationship CSV ---
|
||||
const relCsvPath = path.join(csvDir, 'relations.csv');
|
||||
const relWriter = new BufferedCSVWriter(relCsvPath, 'from,to,type,confidence,reason,step');
|
||||
for (const rel of orderedRelationships(graph, sortOutput)) {
|
||||
await relWriter.addRow(
|
||||
[
|
||||
escapeCSVField(rel.sourceId),
|
||||
escapeCSVField(rel.targetId),
|
||||
escapeCSVField(rel.type),
|
||||
escapeCSVNumber(rel.confidence, 1.0),
|
||||
escapeCSVField(rel.reason),
|
||||
escapeCSVNumber((rel as any).step, 0),
|
||||
].join(','),
|
||||
);
|
||||
}
|
||||
await relWriter.finish();
|
||||
|
||||
// Build result map — only include tables that have rows
|
||||
const nodeFiles = new Map<NodeTableName, { csvPath: string; rows: number }>();
|
||||
const tableMap: [NodeTableName, BufferedCSVWriter][] = [
|
||||
['File', fileWriter],
|
||||
['Folder', folderWriter],
|
||||
['Function', functionWriter],
|
||||
['Class', classWriter],
|
||||
['Interface', interfaceWriter],
|
||||
['Method', methodWriter],
|
||||
['CodeElement', codeElemWriter],
|
||||
['Community', communityWriter],
|
||||
['Process', processWriter],
|
||||
['Section' as NodeTableName, sectionWriter],
|
||||
['Route' as NodeTableName, routeWriter],
|
||||
['Tool' as NodeTableName, toolWriter],
|
||||
['BasicBlock' as NodeTableName, basicBlockWriter],
|
||||
...Array.from(multiLangWriters.entries()).map(
|
||||
([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter],
|
||||
),
|
||||
];
|
||||
for (const [name, writer] of tableMap) {
|
||||
if (writer.rows > 0) {
|
||||
nodeFiles.set(name, {
|
||||
csvPath: path.join(csvDir, `${name.toLowerCase()}.csv`),
|
||||
rows: writer.rows,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Restore original process listener limit
|
||||
process.setMaxListeners(prevMax);
|
||||
|
||||
return { nodeFiles, relCsvPath, relRows: relWriter.rows };
|
||||
};
|
||||
|
|
|
|||
|
|
@ -4,8 +4,6 @@ import { createInterface } from 'readline';
|
|||
import { once } from 'events';
|
||||
import { finished } from 'stream/promises';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import crypto from 'crypto';
|
||||
import lbug from '@ladybugdb/core';
|
||||
import { closeQueryResults } from './query-result-utils.js';
|
||||
import { KnowledgeGraph } from '../graph/types.js';
|
||||
|
|
@ -19,6 +17,8 @@ import {
|
|||
NodeTableName,
|
||||
} from './schema.js';
|
||||
import { streamAllCSVsToDisk } from './csv-generator.js';
|
||||
import type { PdgEmitManifest } from './pdg-emit-sink.js';
|
||||
import { getNodeLabel as deriveNodeLabel, type WriteStreamFactory } from './rel-pair-routing.js';
|
||||
import type { CachedEmbedding } from '../embeddings/types.js';
|
||||
import { extensionManager, type ExtensionEnsureOptions } from './extension-loader.js';
|
||||
import {
|
||||
|
|
@ -28,6 +28,7 @@ import {
|
|||
isWalCorruptionError,
|
||||
openLbugConnection,
|
||||
toNativeSafePath,
|
||||
resolveNativeSafeStorageDir,
|
||||
WAL_RECOVERY_SUGGESTION,
|
||||
waitForWindowsHandleRelease,
|
||||
type LbugConnectionHandle,
|
||||
|
|
@ -48,9 +49,9 @@ import { logger } from '../logger.js';
|
|||
// ---------------------------------------------------------------------------
|
||||
// Relationship CSV splitting — extracted for testability (PR #818)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Factory for creating WriteStreams — injectable for testing. */
|
||||
export type WriteStreamFactory = (filePath: string) => import('fs').WriteStream;
|
||||
// WriteStreamFactory is imported above from rel-pair-routing.ts (its canonical
|
||||
// home) for splitRelCsvByLabelPair's signature; no external code imports it from
|
||||
// here, so it is not re-exported.
|
||||
|
||||
/** Result of splitting the relationship CSV into per-label-pair files. */
|
||||
export interface RelCsvSplitResult {
|
||||
|
|
@ -64,6 +65,15 @@ export interface RelCsvSplitResult {
|
|||
/**
|
||||
* Split a relationship CSV into per-label-pair files on disk.
|
||||
*
|
||||
* @internal RETAINED AS A DIFFERENTIAL ORACLE. As of #2203 U2, production emit
|
||||
* routes relationships to per-pair files directly during the single pass (see
|
||||
* RelPairRouter in `rel-pair-routing.ts`), so this function has NO production
|
||||
* callers — it is kept ONLY so the byte-identity test in
|
||||
* `test/integration/csv-pipeline.test.ts` ("direct per-pair emit matches the
|
||||
* split oracle") can diff the direct-emit output against this proven path. Do
|
||||
* NOT delete it as dead code without also removing that test and accepting the
|
||||
* loss of the byte-identity guard (and likewise `test/unit/rel-csv-split.test.ts`).
|
||||
*
|
||||
* Streams the CSV line-by-line, routing each relationship to a file named
|
||||
* `rel_{fromLabel}_{toLabel}.csv`. Handles backpressure correctly: only one
|
||||
* drain listener per stream at a time, and readline resumes only when ALL
|
||||
|
|
@ -871,6 +881,15 @@ export const loadGraphToLbug = async (
|
|||
repoPath: string,
|
||||
storagePath: string,
|
||||
onProgress?: LbugProgressCallback,
|
||||
/**
|
||||
* Streamed PDG-emit manifest (#2202). When present (streaming was on, full
|
||||
* rebuild), the BasicBlock node CSV + per-pair PDG-edge CSVs it points at
|
||||
* were already flushed to disk during the emit loop; they are merged into the
|
||||
* COPY plan below so they load alongside the structural CSVs. When streaming
|
||||
* was on the in-memory `graph` holds zero BasicBlocks, so `streamAllCSVsToDisk`
|
||||
* emits none — the manifest is the sole source and there is no double-COPY.
|
||||
*/
|
||||
pdgEmitManifest?: PdgEmitManifest,
|
||||
) => {
|
||||
if (!conn) {
|
||||
throw new Error('LadybugDB not initialized. Call initLbug first.');
|
||||
|
|
@ -878,23 +897,57 @@ export const loadGraphToLbug = async (
|
|||
|
||||
const log = onProgress || (() => {});
|
||||
|
||||
let csvDir: string;
|
||||
if (process.platform === 'win32' && /[^\x00-\x7F]/.test(storagePath)) {
|
||||
const hash = crypto.createHash('sha256').update(storagePath).digest('hex').slice(0, 16);
|
||||
csvDir = toNativeSafePath(path.join(os.tmpdir(), `gitnexus-csv-${hash}`));
|
||||
} else {
|
||||
csvDir = path.join(storagePath, 'csv');
|
||||
}
|
||||
// ── #2203 persistence-path profiling ──────────────────────────────────
|
||||
// Mirrors the PROF_SCOPE_RESOLUTION pattern (scope-resolution/pipeline/
|
||||
// run.ts): zero-cost when off — process.hrtime.bigint() is only read under
|
||||
// PROF_LBUG_LOAD=1, and the summary is logged behind the same gate. Fills
|
||||
// the gap that the DB-persistence path is un-timed today (the analyze
|
||||
// "emit" number is the scope-resolution emit bucket, not this COPY path).
|
||||
const PROF = process.env.PROF_LBUG_LOAD === '1';
|
||||
const mark = (): bigint => (PROF ? process.hrtime.bigint() : 0n);
|
||||
const span = (a: bigint, b: bigint): string => (Number(b - a) / 1e6).toFixed(1);
|
||||
const tStart = mark();
|
||||
|
||||
const csvDir = resolveNativeSafeStorageDir(storagePath, 'csv');
|
||||
|
||||
log('Streaming CSVs to disk...');
|
||||
const csvResult = await streamAllCSVsToDisk(graph, repoPath, csvDir);
|
||||
|
||||
// Merge the streamed PDG-emit CSVs (#2202) into the COPY plan so the
|
||||
// BasicBlock node table + per-pair PDG edges (CFG / REACHING_DEF / CDG /
|
||||
// POST_DOMINATE / TAINTED / SANITIZES) load through the SAME node + per-pair
|
||||
// COPY loops as the structural CSVs. The graph held zero BasicBlocks when
|
||||
// streaming, so `streamAllCSVsToDisk` produced none of these — the manifest
|
||||
// is the sole source and there is no double-COPY. Absent ⇒ no-op.
|
||||
if (pdgEmitManifest) {
|
||||
for (const [table, meta] of pdgEmitManifest.nodeFiles) {
|
||||
// A collision means a BasicBlock leaked into the in-memory graph during a
|
||||
// streamed run (streamAllCSVsToDisk then emitted a structural basicblock.csv).
|
||||
// That is a streaming-invariant violation — fail loudly rather than
|
||||
// silently overwrite one CSV with the other and drop its rows (#2202 review #3).
|
||||
if (csvResult.nodeFiles.has(table)) {
|
||||
throw new Error(
|
||||
`Streaming PDG manifest collides with a structural node CSV for "${table}" — ` +
|
||||
`the in-memory graph should hold zero ${table} nodes when streaming. ` +
|
||||
`A ${table} node leaked into the graph during a streamed emit.`,
|
||||
);
|
||||
}
|
||||
csvResult.nodeFiles.set(table, meta);
|
||||
}
|
||||
for (const [pairKey, meta] of pdgEmitManifest.relsByPair) {
|
||||
if (csvResult.relsByPair.has(pairKey)) {
|
||||
throw new Error(
|
||||
`Streaming PDG manifest collides with a structural relationship CSV for pair ` +
|
||||
`"${pairKey}" — a PDG edge leaked into the in-memory graph during a streamed emit.`,
|
||||
);
|
||||
}
|
||||
csvResult.relsByPair.set(pairKey, meta);
|
||||
csvResult.totalValidRels += meta.rows;
|
||||
}
|
||||
}
|
||||
const tCsv = mark();
|
||||
|
||||
const validTables = new Set<string>(NODE_TABLES as readonly string[]);
|
||||
const getNodeLabel = (nodeId: string): string => {
|
||||
if (nodeId.startsWith('comm_')) return 'Community';
|
||||
if (nodeId.startsWith('proc_')) return 'Process';
|
||||
return nodeId.split(':')[0];
|
||||
};
|
||||
|
||||
// Bulk COPY all node CSVs (sequential — LadybugDB allows only one write txn at a time)
|
||||
const nodeFiles = [...csvResult.nodeFiles.entries()];
|
||||
|
|
@ -924,37 +977,32 @@ export const loadGraphToLbug = async (
|
|||
}
|
||||
}
|
||||
|
||||
// Bulk COPY relationships — split by FROM→TO label pair (LadybugDB requires it)
|
||||
const { relHeader, relsByPairMeta, pairWriteStreams, skippedRels, totalValidRels } =
|
||||
await splitRelCsvByLabelPair(csvResult.relCsvPath, csvDir, validTables, getNodeLabel);
|
||||
const tCopyNodes = mark();
|
||||
|
||||
// Close all per-pair write streams before COPY. `stream/promises.finished`
|
||||
// resolves on the stream's 'finish' event and rejects on 'error' — replaces
|
||||
// a hand-rolled promisification with the stdlib primitive.
|
||||
await Promise.all(
|
||||
Array.from(pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
// Bulk COPY relationships. They were already routed to per-FROM→TO-label-pair
|
||||
// files during the emit pass (#2203 U2) — there is no monolithic relations.csv
|
||||
// to re-read/re-split here; we COPY each pair file directly.
|
||||
const { relsByPair, relHeader, skippedRels, totalValidRels } = csvResult;
|
||||
let tCopyRels = tCopyNodes;
|
||||
let tFallback = tCopyNodes;
|
||||
|
||||
const insertedRels = totalValidRels;
|
||||
const warnings: string[] = [];
|
||||
if (insertedRels > 0) {
|
||||
log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPairMeta.size} types`);
|
||||
log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPair.size} types`);
|
||||
|
||||
let pairIdx = 0;
|
||||
let failedPairEdges = 0;
|
||||
const failedPairCsvPaths = new Set<string>();
|
||||
|
||||
for (const [pairKey, { csvPath: pairCsvPath, rows }] of relsByPairMeta) {
|
||||
for (const [pairKey, { csvPath: pairCsvPath, rows }] of relsByPair) {
|
||||
pairIdx++;
|
||||
const [fromLabel, toLabel] = pairKey.split('|');
|
||||
const normalizedPath = normalizeCopyPath(pairCsvPath);
|
||||
const copyQuery = `COPY ${REL_TABLE_NAME} FROM "${normalizedPath}" (from="${fromLabel}", to="${toLabel}", HEADER=true, ESCAPE='"', DELIM=',', QUOTE='"', PARALLEL=false, auto_detect=false)`;
|
||||
|
||||
if (pairIdx % 5 === 0 || rows > 1000) {
|
||||
log(`Loading edges: ${pairIdx}/${relsByPairMeta.size} types (${fromLabel} -> ${toLabel})`);
|
||||
log(`Loading edges: ${pairIdx}/${relsByPair.size} types (${fromLabel} -> ${toLabel})`);
|
||||
}
|
||||
|
||||
try {
|
||||
|
|
@ -980,6 +1028,7 @@ export const loadGraphToLbug = async (
|
|||
} catch {}
|
||||
}
|
||||
}
|
||||
tCopyRels = mark();
|
||||
|
||||
if (failedPairCsvPaths.size > 0) {
|
||||
log(`Inserting ${failedPairEdges} edges individually (missing schema pairs)`);
|
||||
|
|
@ -999,15 +1048,14 @@ export const loadGraphToLbug = async (
|
|||
} catch {}
|
||||
}
|
||||
if (allLines.length > 1) {
|
||||
await fallbackRelationshipInserts(allLines, validTables, getNodeLabel);
|
||||
await fallbackRelationshipInserts(allLines, validTables, deriveNodeLabel);
|
||||
}
|
||||
}
|
||||
tFallback = mark();
|
||||
}
|
||||
|
||||
// Cleanup all CSVs
|
||||
try {
|
||||
await fs.unlink(csvResult.relCsvPath);
|
||||
} catch {}
|
||||
// Cleanup all CSVs (per-pair rel files are unlinked in the COPY loop above;
|
||||
// the remaining sweep below catches node CSVs + any leftover pair files).
|
||||
for (const [, { csvPath }] of csvResult.nodeFiles) {
|
||||
try {
|
||||
await fs.unlink(csvPath);
|
||||
|
|
@ -1025,6 +1073,18 @@ export const loadGraphToLbug = async (
|
|||
await fs.rmdir(csvDir);
|
||||
} catch {}
|
||||
|
||||
if (PROF) {
|
||||
const tEnd = mark();
|
||||
let totalNodeRows = 0;
|
||||
for (const [, { rows }] of csvResult.nodeFiles) totalNodeRows += rows;
|
||||
logger.warn(
|
||||
`[lbug-load prof] csv-emit=${span(tStart, tCsv)}ms ` +
|
||||
`copy-nodes=${span(tCsv, tCopyNodes)}ms copy-rels=${span(tCopyNodes, tCopyRels)}ms ` +
|
||||
`fallback=${span(tCopyRels, tFallback)}ms total=${span(tStart, tEnd)}ms ` +
|
||||
`(${totalNodeRows} nodes, ${insertedRels} rels)`,
|
||||
);
|
||||
}
|
||||
|
||||
return { success: true, insertedRels, skippedRels, warnings };
|
||||
};
|
||||
|
||||
|
|
|
|||
|
|
@ -189,6 +189,36 @@ export function toNativeSafePath(p: string): string {
|
|||
return p;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the on-disk CSV staging dir for `<storagePath>/<subdir>`, applying the
|
||||
* same ASCII-safe relocation `toNativeSafePath` enables: on Windows with a
|
||||
* non-ASCII storage path, LadybugDB's bulk COPY cannot open files under that
|
||||
* path, so the dir is relocated under `os.tmpdir()`. Shared by the structural
|
||||
* `csv/` dir and the streaming `pdg-csv/` dir (#2202) so the two can never
|
||||
* diverge on platform handling; the `gitnexus-<subdir>-` prefix keeps their tmp
|
||||
* locations distinct and recognizable.
|
||||
*
|
||||
* The relocated dir is created with `fs.mkdtempSync` (a unique, mode-0700,
|
||||
* guaranteed-not-pre-existing suffix) rather than a deterministic
|
||||
* `gitnexus-<subdir>-<hash>` name. A predictable name in the world-readable OS
|
||||
* temp dir is information-disclosure-prone and pre-plantable
|
||||
* (CWE-377/378 / CodeQL `js/insecure-temporary-file`); mkdtemp's random suffix
|
||||
* is the documented mitigation and is what reaches the streaming sink's
|
||||
* `fs.openSync`. The non-Windows / ASCII path stays a pure `path.join` (no dir
|
||||
* created) and is byte-identical to before.
|
||||
*/
|
||||
export function resolveNativeSafeStorageDir(storagePath: string, subdir: string): string {
|
||||
if (process.platform === 'win32' && NON_ASCII_RE.test(storagePath)) {
|
||||
// 8.3-shorten the tmpdir base first (a non-ASCII Windows *profile* path can
|
||||
// make os.tmpdir() itself non-ASCII), THEN mkdtemp so the returned path —
|
||||
// the one that flows into fs.openSync — is provably mkdtemp-sourced (random,
|
||||
// exclusive) and clears the insecure-temp-file dataflow.
|
||||
const base = toNativeSafePath(os.tmpdir());
|
||||
return fsSync.mkdtempSync(path.join(base, `gitnexus-${subdir}-`));
|
||||
}
|
||||
return path.join(storagePath, subdir);
|
||||
}
|
||||
|
||||
/**
|
||||
* Shared configuration for `@ladybugdb/core` `Database` construction.
|
||||
*
|
||||
|
|
|
|||
395
gitnexus/src/core/lbug/pdg-emit-sink.ts
Normal file
395
gitnexus/src/core/lbug/pdg-emit-sink.ts
Normal file
|
|
@ -0,0 +1,395 @@
|
|||
/**
|
||||
* Streaming PDG graph-emit sink (issue #2202).
|
||||
*
|
||||
* The PDG emit loop (`scope-resolution/pipeline/run.ts`, the `--pdg` block)
|
||||
* materializes BasicBlock nodes + intra-file PDG edges (CFG / REACHING_DEF /
|
||||
* CDG / POST_DOMINATE / TAINTED / SANITIZES) into the in-memory
|
||||
* `KnowledgeGraph`. At full-kernel scale that layer dominates peak RSS
|
||||
* (~7 GB at 511K BasicBlocks; ~100 GB extrapolated to the full kernel → OOM).
|
||||
*
|
||||
* `PdgEmitSink` is a write-routing façade over the real graph: the emit
|
||||
* functions are write-only and compute every edge endpoint by deterministic id
|
||||
* (audited — no read-back), so the sink can route BasicBlock node rows and PDG
|
||||
* edge rows straight to bounded CSV-on-disk writers and **never store them**.
|
||||
* The graph's resident size stops growing with the PDG layer → peak RSS becomes
|
||||
* O(chunk buffer), not O(graph). Everything else (structural nodes/edges, the
|
||||
* whole-program M4 TAINT_PATH edges) is delegated to the real graph unchanged.
|
||||
*
|
||||
* Why synchronous writers? The whole PDG emit (`runScopeResolution` and its
|
||||
* per-file loop) is synchronous — there is no `await` point to drain an async
|
||||
* stream, so a `BufferedCSVWriter` (Node `WriteStream`) would accumulate
|
||||
* unwritten chunks in process memory across millions of rows, defeating the RSS
|
||||
* bound. `fs.writeSync` goes straight to the OS; resident memory is bounded to
|
||||
* one `chunkRows` buffer. This mirrors the sync-shard pattern in
|
||||
* `storage/parsedfile-store.ts`.
|
||||
*
|
||||
* Byte-identity (issue acceptance): the sink reuses the SAME shared row
|
||||
* builders (`buildBasicBlockRow`, `buildRelRow`) and label derivation
|
||||
* (`getNodeLabel`) as `streamAllCSVsToDisk`, so the streamed CSV line SET is
|
||||
* identical to the whole-graph emit's, and the bulk COPY loads the same rows →
|
||||
* the persisted graph is SET-identical and DB-identical. The guarantee is
|
||||
* set-level, not byte-level on the CSV file: the sink streams rows in emit
|
||||
* order and does NOT re-sort them under `GITNEXUS_SORT_GRAPH_OUTPUT`, so a
|
||||
* streamed CSV file is not necessarily byte-for-byte equal to the sorted
|
||||
* whole-graph CSV — but the row set, and therefore the DB outcome, is. (The
|
||||
* streamed CSVs are deleted right after the COPY, so their on-disk byte order
|
||||
* is never observed.) Cross-pass dedup is done upstream, per FILE, in the emit
|
||||
* loop (`run.ts` skips a file whose PDG already streamed) rather than in the
|
||||
* sink, because a file can be PDG-emitted in more than one language pass (a
|
||||
* `.ts` module imported by a `.vue` SFC is emitted in both the TypeScript and
|
||||
* Vue context passes over the same worker-built CFG) and a sink-level per-id
|
||||
* dedup set would retain every id → O(total ids) memory, defeating the
|
||||
* O(chunk) RSS bound (#2202 review #1). The differential fingerprint test
|
||||
* (issue #2202 U6) and the Vue+TS cross-pass integration test guard the set.
|
||||
*/
|
||||
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import type { GraphNode, GraphRelationship, RelationshipType } from 'gitnexus-shared';
|
||||
import type { KnowledgeGraph } from '../graph/types.js';
|
||||
import {
|
||||
BASICBLOCK_CSV_HEADER,
|
||||
REL_CSV_HEADER,
|
||||
buildBasicBlockRow,
|
||||
buildRelRow,
|
||||
} from './csv-generator.js';
|
||||
import { getNodeLabel } from './rel-pair-routing.js';
|
||||
import { NODE_TABLES, type NodeTableName } from './schema.js';
|
||||
|
||||
/**
|
||||
* PDG edge types streamed per-file (all intra-block BasicBlock→BasicBlock).
|
||||
* `TAINT_PATH` is intentionally excluded — it is the whole-program M4 edge
|
||||
* (Function→Function), computed in a separate post-resolution phase over the
|
||||
* complete CALLS graph, and stays in the in-memory graph (it is small and is
|
||||
* persisted by the normal whole-graph emit).
|
||||
*/
|
||||
const PDG_EDGE_TYPES: ReadonlySet<RelationshipType> = new Set<RelationshipType>([
|
||||
'CFG',
|
||||
'REACHING_DEF',
|
||||
'CDG',
|
||||
'POST_DOMINATE',
|
||||
'TAINTED',
|
||||
'SANITIZES',
|
||||
]);
|
||||
|
||||
/** Default streamed-write buffer (rows). Matches the whole-graph emit's
|
||||
* `FLUSH_EVERY` order of magnitude; overridable via `GITNEXUS_PDG_EMIT_CHUNK_SIZE`. */
|
||||
export const DEFAULT_PDG_EMIT_CHUNK_ROWS = 500;
|
||||
|
||||
/**
|
||||
* Synchronous buffered CSV writer. Buffers up to `chunkRows` rows, then issues
|
||||
* one `fs.writeSync` straight to the OS (no in-process stream buffer). Header
|
||||
* is written into the buffer at construction and is NOT counted in `rows`
|
||||
* (matching `BufferedCSVWriter` semantics, so manifest row counts line up).
|
||||
*/
|
||||
class SyncCsvWriter {
|
||||
private fd: number;
|
||||
private buf: string[] = [];
|
||||
private readonly chunkRows: number;
|
||||
rows = 0;
|
||||
/**
|
||||
* First IO error this writer hit (a `fs.writeSync` short-write loop throwing
|
||||
* on e.g. disk-full). Once poisoned the writer refuses further rows and
|
||||
* skips its final flush; the sink surfaces it from {@link PdgEmitSink.finalize}
|
||||
* so a truncated CSV is never handed to the bulk COPY (#2202 review #4). A
|
||||
* streamed-write failure is an IO fault, not the CFG-logic error that the
|
||||
* emit loop's per-file try/catch is built to swallow — poisoning routes it
|
||||
* past that catch to a loud failure.
|
||||
*/
|
||||
poison: unknown | undefined = undefined;
|
||||
|
||||
constructor(
|
||||
readonly csvPath: string,
|
||||
header: string,
|
||||
chunkRows: number,
|
||||
) {
|
||||
// Guard a 0/negative buffer: the flush modulo would never fire and `buf`
|
||||
// would grow unbounded, defeating the whole point of streaming.
|
||||
this.chunkRows = Math.max(1, chunkRows);
|
||||
// Exclusive create (O_EXCL): the streamed-CSV dir is wiped + recreated fresh
|
||||
// by the PdgEmitSink constructor before any writer opens a file, so the path
|
||||
// never pre-exists — 'wx' both matches that invariant and refuses to follow
|
||||
// a pre-planted symlink at the path (CWE-377 / CodeQL js/insecure-temporary-file).
|
||||
this.fd = fs.openSync(csvPath, 'wx');
|
||||
this.buf.push(header);
|
||||
}
|
||||
|
||||
addRow(row: string): void {
|
||||
// A poisoned writer is dead — stop buffering so memory can't grow on a
|
||||
// writer whose fd is already in a bad state; finalize will report the fault.
|
||||
if (this.poison !== undefined) return;
|
||||
this.buf.push(row);
|
||||
this.rows++;
|
||||
// Flush on DATA-row count, not buffer length: the header occupies buf[0]
|
||||
// until the first flush, so a `buf.length >= chunkRows` test would fire one
|
||||
// row early on the first chunk. Counting rows makes every flush exactly
|
||||
// `chunkRows` rows.
|
||||
if (this.rows % this.chunkRows === 0) this.flushOrPoison();
|
||||
}
|
||||
|
||||
/** Flush, recording (and re-throwing) any IO error as poison. Re-throwing
|
||||
* lets the immediate caller log the per-file failure; the persisted `poison`
|
||||
* is the backstop that makes finalize fail loudly even when that throw is
|
||||
* swallowed by the emit loop's CFG try/catch. */
|
||||
private flushOrPoison(): void {
|
||||
try {
|
||||
this.flush();
|
||||
} catch (e) {
|
||||
this.poison ??= e;
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
private flush(): void {
|
||||
if (this.buf.length === 0) return;
|
||||
const data = Buffer.from(this.buf.join('\n') + '\n', 'utf8');
|
||||
// fs.writeSync can return a short byte count; loop until the whole buffer
|
||||
// lands so a partial write never truncates a CSV row mid-field.
|
||||
let offset = 0;
|
||||
while (offset < data.length) {
|
||||
offset += fs.writeSync(this.fd, data, offset, data.length - offset);
|
||||
}
|
||||
this.buf.length = 0;
|
||||
}
|
||||
|
||||
/** Flush remaining rows (unless already poisoned) and close the fd. Never
|
||||
* throws: a final-flush IO error is recorded as poison and the fd is still
|
||||
* closed, so a write error neither leaks an fd nor escapes here — the sink
|
||||
* reads {@link poison} after closing every writer and fails loudly then. */
|
||||
close(): void {
|
||||
try {
|
||||
if (this.poison === undefined) this.flush();
|
||||
} catch (e) {
|
||||
this.poison ??= e;
|
||||
} finally {
|
||||
try {
|
||||
fs.closeSync(this.fd);
|
||||
} catch {
|
||||
/* fd may already be invalid after an IO fault — nothing to recover */
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* COPY manifest produced by {@link PdgEmitSink.finalize}. Shaped to merge
|
||||
* directly into `StreamedCSVResult` so `loadGraphToLbug` COPYs the streamed
|
||||
* PDG CSVs through the same per-table / per-pair loops as the structural CSVs.
|
||||
* Paths are absolute, so persistence needs no dir recomputation.
|
||||
*/
|
||||
export interface PdgEmitManifest {
|
||||
/** Node-table CSVs (only `BasicBlock` today). */
|
||||
readonly nodeFiles: Map<NodeTableName, { csvPath: string; rows: number }>;
|
||||
/** pairKey (`From|To`) → per-pair edge CSV. */
|
||||
readonly relsByPair: Map<string, { csvPath: string; rows: number }>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write-routing graph façade. Construct one per analyze run, thread it into the
|
||||
* per-language `runScopeResolution` calls in place of the real graph during the
|
||||
* `--pdg` emit, then {@link finalize} once after the last language.
|
||||
*/
|
||||
export class PdgEmitSink implements KnowledgeGraph {
|
||||
private readonly validTables: Set<string>;
|
||||
private bbWriter: SyncCsvWriter | undefined;
|
||||
/** pairKey (`From|To`) → writer. PDG edges are all `BasicBlock|BasicBlock`,
|
||||
* but the map keeps the sink general and the manifest pair-keyed. */
|
||||
private readonly relWriters = new Map<string, SyncCsvWriter>();
|
||||
private finalized = false;
|
||||
/**
|
||||
* First writer-construction failure (a `fs.openSync` throwing on e.g. EMFILE
|
||||
* — out of file descriptors). The failure happens inside the `SyncCsvWriter`
|
||||
* constructor before a writer object exists to carry poison, so it is held
|
||||
* here at the sink level and folded into the {@link finalize} error check.
|
||||
* Like an in-flight write fault, an open failure mid-emit would otherwise be
|
||||
* swallowed by the emit loop's per-file try/catch and silently drop the rest
|
||||
* of that file's rows (#2202 review #4/#6).
|
||||
*/
|
||||
private openFailure: unknown | undefined = undefined;
|
||||
// NOTE on dedup: the same file can be PDG-emitted in more than one language
|
||||
// pass (e.g. a `.ts` module imported by a `.vue` SFC is emitted in both the
|
||||
// TypeScript pass and the Vue context pass over the same worker-built
|
||||
// `cfgSideChannel`). The in-memory graph dedups that by id (first-writer-wins);
|
||||
// this sink does NOT — to keep peak memory O(write buffer) rather than
|
||||
// O(total ids), cross-pass dedup is done upstream, per FILE, in the emit loop
|
||||
// (`run.ts` skips a file whose PDG already streamed via `pdgEmittedFiles`).
|
||||
// The sink therefore receives each id exactly once and is a faithful
|
||||
// pass-through; it must not be fed duplicate ids.
|
||||
|
||||
constructor(
|
||||
private readonly real: KnowledgeGraph,
|
||||
private readonly pdgCsvDir: string,
|
||||
private readonly chunkRows: number = DEFAULT_PDG_EMIT_CHUNK_ROWS,
|
||||
) {
|
||||
this.validTables = new Set<string>(NODE_TABLES as readonly string[]);
|
||||
// Clear any streamed CSVs left by a previous (possibly crashed) run so a
|
||||
// later COPY never picks up stale rows.
|
||||
fs.rmSync(pdgCsvDir, { recursive: true, force: true });
|
||||
fs.mkdirSync(pdgCsvDir, { recursive: true });
|
||||
}
|
||||
|
||||
// ── routed writes ──────────────────────────────────────────────────────────
|
||||
|
||||
addNode(node: GraphNode): void {
|
||||
if (node.label === 'BasicBlock') {
|
||||
if (this.bbWriter === undefined) {
|
||||
try {
|
||||
this.bbWriter = new SyncCsvWriter(
|
||||
path.join(this.pdgCsvDir, 'basicblock.csv'),
|
||||
BASICBLOCK_CSV_HEADER,
|
||||
this.chunkRows,
|
||||
);
|
||||
} catch (e) {
|
||||
this.openFailure ??= e;
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
this.bbWriter.addRow(buildBasicBlockRow(node));
|
||||
return;
|
||||
}
|
||||
this.real.addNode(node);
|
||||
}
|
||||
|
||||
addRelationship(relationship: GraphRelationship): void {
|
||||
if (PDG_EDGE_TYPES.has(relationship.type)) {
|
||||
const fromLabel = getNodeLabel(relationship.sourceId);
|
||||
const toLabel = getNodeLabel(relationship.targetId);
|
||||
// Skip edges whose endpoint labels are not valid node tables — mirrors
|
||||
// `RelPairRouter` exactly so the streamed set matches the whole-graph set.
|
||||
if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) return;
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
let writer = this.relWriters.get(pairKey);
|
||||
if (writer === undefined) {
|
||||
try {
|
||||
writer = new SyncCsvWriter(
|
||||
path.join(this.pdgCsvDir, `rel_${fromLabel}_${toLabel}.csv`),
|
||||
REL_CSV_HEADER,
|
||||
this.chunkRows,
|
||||
);
|
||||
} catch (e) {
|
||||
this.openFailure ??= e;
|
||||
throw e;
|
||||
}
|
||||
this.relWriters.set(pairKey, writer);
|
||||
}
|
||||
writer.addRow(buildRelRow(relationship));
|
||||
return;
|
||||
}
|
||||
this.real.addRelationship(relationship);
|
||||
}
|
||||
|
||||
/** Flush + close every streamed writer and return the COPY manifest. Every
|
||||
* fd is closed even when a writer is poisoned (its `close` never throws); any
|
||||
* IO fault — an in-flight write that poisoned a writer, a final-flush failure,
|
||||
* or a writer-open failure (EMFILE) — is surfaced loudly here so a disk-full
|
||||
* / out-of-fds run never hands a truncated CSV to the bulk COPY (#2202 review
|
||||
* #4). The emit loop's per-file try/catch swallows the synchronous throw, so
|
||||
* this poison check is the backstop that turns a silent partial manifest into
|
||||
* a hard failure. */
|
||||
finalize(): PdgEmitManifest {
|
||||
if (this.finalized) throw new Error('PdgEmitSink.finalize() called twice');
|
||||
this.finalized = true;
|
||||
|
||||
const errors: unknown[] = [];
|
||||
if (this.openFailure !== undefined) errors.push(this.openFailure);
|
||||
|
||||
const nodeFiles = new Map<NodeTableName, { csvPath: string; rows: number }>();
|
||||
if (this.bbWriter !== undefined) {
|
||||
this.bbWriter.close();
|
||||
if (this.bbWriter.poison !== undefined) errors.push(this.bbWriter.poison);
|
||||
nodeFiles.set('BasicBlock' as NodeTableName, {
|
||||
csvPath: this.bbWriter.csvPath,
|
||||
rows: this.bbWriter.rows,
|
||||
});
|
||||
}
|
||||
|
||||
const relsByPair = new Map<string, { csvPath: string; rows: number }>();
|
||||
for (const [pairKey, writer] of this.relWriters) {
|
||||
writer.close();
|
||||
if (writer.poison !== undefined) errors.push(writer.poison);
|
||||
relsByPair.set(pairKey, { csvPath: writer.csvPath, rows: writer.rows });
|
||||
}
|
||||
|
||||
if (errors.length > 0) {
|
||||
const first = errors[0];
|
||||
throw new Error(
|
||||
`PdgEmitSink: ${errors.length} streamed CSV writer(s) hit an IO error ` +
|
||||
`(disk-full / out-of-fds) during the emit — the persisted graph would ` +
|
||||
`be truncated, so the run is failed rather than COPYing a partial CSV: ${
|
||||
first instanceof Error ? first.message : String(first)
|
||||
}`,
|
||||
);
|
||||
}
|
||||
|
||||
return { nodeFiles, relsByPair };
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort fd release for the error path — when a language pass throws
|
||||
* before {@link finalize} runs, the caller's `finally` calls this so the
|
||||
* BasicBlock + per-pair fds never leak. Idempotent with finalize via the
|
||||
* `finalized` flag; close errors are swallowed because the run is already
|
||||
* failing.
|
||||
*/
|
||||
close(): void {
|
||||
if (this.finalized) return;
|
||||
this.finalized = true;
|
||||
try {
|
||||
this.bbWriter?.close();
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
for (const writer of this.relWriters.values()) {
|
||||
try {
|
||||
writer.close();
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── delegated reads / non-PDG mutations ─────────────────────────────────────
|
||||
// The PDG emit functions never call these on the routed graph, but the
|
||||
// façade implements the full KnowledgeGraph surface so it is a drop-in for
|
||||
// the emit target and any non-PDG write transparently reaches the real graph.
|
||||
|
||||
get nodes(): GraphNode[] {
|
||||
return this.real.nodes;
|
||||
}
|
||||
get relationships(): GraphRelationship[] {
|
||||
return this.real.relationships;
|
||||
}
|
||||
iterNodes(): IterableIterator<GraphNode> {
|
||||
return this.real.iterNodes();
|
||||
}
|
||||
iterRelationships(): IterableIterator<GraphRelationship> {
|
||||
return this.real.iterRelationships();
|
||||
}
|
||||
iterRelationshipsByType(type: RelationshipType): IterableIterator<GraphRelationship> {
|
||||
return this.real.iterRelationshipsByType(type);
|
||||
}
|
||||
forEachNode(fn: (node: GraphNode) => void): void {
|
||||
this.real.forEachNode(fn);
|
||||
}
|
||||
forEachRelationship(fn: (rel: GraphRelationship) => void): void {
|
||||
this.real.forEachRelationship(fn);
|
||||
}
|
||||
getNode(id: string): GraphNode | undefined {
|
||||
return this.real.getNode(id);
|
||||
}
|
||||
get nodeCount(): number {
|
||||
return this.real.nodeCount;
|
||||
}
|
||||
get relationshipCount(): number {
|
||||
return this.real.relationshipCount;
|
||||
}
|
||||
removeNode(nodeId: string): boolean {
|
||||
return this.real.removeNode(nodeId);
|
||||
}
|
||||
removeNodesByFile(filePath: string): number {
|
||||
return this.real.removeNodesByFile(filePath);
|
||||
}
|
||||
removeRelationship(relationshipId: string): boolean {
|
||||
return this.real.removeRelationship(relationshipId);
|
||||
}
|
||||
}
|
||||
159
gitnexus/src/core/lbug/rel-pair-routing.ts
Normal file
159
gitnexus/src/core/lbug/rel-pair-routing.ts
Normal file
|
|
@ -0,0 +1,159 @@
|
|||
/**
|
||||
* Relationship per-label-pair routing (#2203 U2).
|
||||
*
|
||||
* LadybugDB's bulk `COPY` into the single `CodeRelation` rel table requires a
|
||||
* separate CSV per FROM→TO node-label pair (the `from=`/`to=` COPY params).
|
||||
* Historically the emit pass wrote one monolithic `relations.csv`, which
|
||||
* `loadGraphToLbug` then RE-READ line-by-line (regex per edge) and re-split
|
||||
* into per-pair files — writing and reading the entire ~1M-edge set twice.
|
||||
*
|
||||
* This router lets the single emit pass route each edge to its per-pair file
|
||||
* directly, so the monolithic write + re-read + per-edge regex are all gone.
|
||||
* The label-derivation + validTables filtering + per-pair-file format here match
|
||||
* the legacy `splitRelCsvByLabelPair`, so the per-pair files are byte-identical
|
||||
* for all quote-free ids — see the differential test in
|
||||
* `test/integration/csv-pipeline.test.ts`. ONE intentional divergence: this
|
||||
* router derives the label from the RAW id, while the oracle re-derives it via a
|
||||
* regex over the ESCAPED row — so for an id containing a `"` the router is the
|
||||
* more-correct path (it routes the edge to the right pair; the oracle's regex
|
||||
* mis-buckets or drops it). `splitRelCsvByLabelPair` is retained as the
|
||||
* differential oracle (the quote-in-id divergence is asserted explicitly).
|
||||
*
|
||||
* Backpressure: at most one stream is awaited at a time (the caller routes
|
||||
* edges sequentially and awaits the returned drain promise before the next),
|
||||
* mirroring the legacy split's `for await` invariant. The hot path (existing
|
||||
* pair, no backpressure) returns `void` — no microtask per edge.
|
||||
*/
|
||||
import path from 'path';
|
||||
import { createWriteStream, type WriteStream } from 'fs';
|
||||
import { once } from 'events';
|
||||
import { finished } from 'stream/promises';
|
||||
|
||||
/** Injectable for tests (backpressure/error simulation), mirroring split. */
|
||||
export type WriteStreamFactory = (filePath: string) => WriteStream;
|
||||
|
||||
/**
|
||||
* Derive a node's table label from its graph id. Matches the legacy
|
||||
* `getNodeLabel` that lived inline in `loadGraphToLbug`:
|
||||
* - `comm_*` → Community
|
||||
* - `proc_*` → Process
|
||||
* - otherwise the prefix before the first `:` (e.g. `Function:…` → Function)
|
||||
*/
|
||||
export const getNodeLabel = (nodeId: string): string => {
|
||||
if (nodeId.startsWith('comm_')) return 'Community';
|
||||
if (nodeId.startsWith('proc_')) return 'Process';
|
||||
return nodeId.split(':')[0];
|
||||
};
|
||||
|
||||
export interface RelPairMeta {
|
||||
csvPath: string;
|
||||
rows: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Routes already-escaped relationship CSV rows to per-FROM→TO-label-pair
|
||||
* files. Filters edges whose endpoint labels are not valid node tables
|
||||
* (counted as `skipped`), exactly as the legacy split did.
|
||||
*/
|
||||
export class RelPairRouter {
|
||||
/** pairKey (`From|To`) → { csvPath, rows } */
|
||||
readonly byPair = new Map<string, RelPairMeta>();
|
||||
private readonly streams = new Map<string, WriteStream>();
|
||||
skipped = 0;
|
||||
total = 0;
|
||||
|
||||
private streamError: Error | null = null;
|
||||
private readonly abort = new AbortController();
|
||||
|
||||
constructor(
|
||||
private readonly csvDir: string,
|
||||
private readonly header: string,
|
||||
private readonly validTables: Set<string>,
|
||||
private readonly wsFactory: WriteStreamFactory = (p) => createWriteStream(p, 'utf-8'),
|
||||
) {}
|
||||
|
||||
private markError = (err: Error): void => {
|
||||
this.streamError ??= err;
|
||||
this.abort.abort(err);
|
||||
};
|
||||
|
||||
/**
|
||||
* The first stream error observed, if any. Lets the emit caller rethrow the
|
||||
* real error (EMFILE / disk-full) instead of the generic `AbortError` that a
|
||||
* pending `once(ws,'drain',{signal})` rejects with when the abort fires —
|
||||
* mirroring the retained `splitRelCsvByLabelPair`'s `throw streamError ?? err`.
|
||||
*/
|
||||
get lastError(): Error | null {
|
||||
return this.streamError;
|
||||
}
|
||||
|
||||
/**
|
||||
* Route one already-escaped CSV row (no trailing newline) to its pair file.
|
||||
* Returns `void` on the synchronous hot path; a `Promise<void>` only when a
|
||||
* stream signals backpressure (or a new pair's header does) — the caller
|
||||
* awaits the promise before routing the next edge.
|
||||
*/
|
||||
route(fromId: string, toId: string, row: string): void | Promise<void> {
|
||||
if (this.streamError) throw this.streamError;
|
||||
|
||||
const fromLabel = getNodeLabel(fromId);
|
||||
const toLabel = getNodeLabel(toId);
|
||||
if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) {
|
||||
this.skipped++;
|
||||
return;
|
||||
}
|
||||
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
const ws = this.streams.get(pairKey);
|
||||
if (ws === undefined) {
|
||||
// First edge for this pair: open the stream, write header + row.
|
||||
return this.openAndWrite(pairKey, fromLabel, toLabel, row);
|
||||
}
|
||||
|
||||
this.byPair.get(pairKey)!.rows++;
|
||||
this.total++;
|
||||
if (!ws.write(row + '\n')) {
|
||||
return once(ws, 'drain', { signal: this.abort.signal }).then(() => undefined);
|
||||
}
|
||||
}
|
||||
|
||||
private async openAndWrite(
|
||||
pairKey: string,
|
||||
fromLabel: string,
|
||||
toLabel: string,
|
||||
row: string,
|
||||
): Promise<void> {
|
||||
const csvPath = path.join(this.csvDir, `rel_${fromLabel}_${toLabel}.csv`);
|
||||
const ws = this.wsFactory(csvPath);
|
||||
ws.on('error', this.markError);
|
||||
this.streams.set(pairKey, ws);
|
||||
this.byPair.set(pairKey, { csvPath, rows: 1 });
|
||||
this.total++;
|
||||
if (!ws.write(this.header + '\n')) {
|
||||
await once(ws, 'drain', { signal: this.abort.signal });
|
||||
}
|
||||
if (!ws.write(row + '\n')) {
|
||||
await once(ws, 'drain', { signal: this.abort.signal });
|
||||
}
|
||||
}
|
||||
|
||||
/** Flush + close every pair stream. Rejects if any stream errored. */
|
||||
async close(): Promise<void> {
|
||||
if (this.streamError) {
|
||||
this.destroy();
|
||||
throw this.streamError;
|
||||
}
|
||||
await Promise.all(
|
||||
Array.from(this.streams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
if (this.streamError) throw this.streamError;
|
||||
}
|
||||
|
||||
/** Tear down all streams (no flush) — used on the error path. */
|
||||
destroy(): void {
|
||||
for (const ws of this.streams.values()) ws.destroy();
|
||||
}
|
||||
}
|
||||
|
|
@ -61,6 +61,7 @@ import {
|
|||
} from './ingestion/taint/interproc-solver.js';
|
||||
import { DEFAULT_PDG_MAX_INTERPROC_EDGES } from './ingestion/taint/interproc-emit.js';
|
||||
import { taintModelVersion } from './ingestion/taint/typescript-model.js';
|
||||
import { parseTruthyEnv, parsePositiveIntEnv } from './ingestion/utils/env.js';
|
||||
import { computeFileHashes, diffFileHashes } from '../storage/file-hash.js';
|
||||
import {
|
||||
extractChangedSubgraph,
|
||||
|
|
@ -170,6 +171,19 @@ export interface AnalyzeOptions {
|
|||
pdgMaxInterprocFindings?: number;
|
||||
pdgMaxInterprocHops?: number;
|
||||
pdgMaxInterprocEdges?: number;
|
||||
/**
|
||||
* Stream the BasicBlock + intra-file PDG-edge layer to CSV-on-disk during the
|
||||
* emit loop instead of materializing it in the in-memory graph, bounding peak
|
||||
* RSS to O(chunk) for full-kernel-scale repos (#2202). Only engages on a full
|
||||
* rebuild — `resolveStreamPdgEmit` additionally requires `force === true`
|
||||
* (the pre-pipeline guarantee of a full rebuild). May also be enabled via
|
||||
* `GITNEXUS_STREAM_PDG_EMIT`. Memory-only; byte-identical output; not stamped
|
||||
* into `RepoMeta.pdg`. */
|
||||
streamPdgEmit?: boolean;
|
||||
/** Streamed PDG-emit write buffer (rows). `undefined` ⇒
|
||||
* `DEFAULT_PDG_EMIT_CHUNK_ROWS`. May also be set via
|
||||
* `GITNEXUS_PDG_EMIT_CHUNK_SIZE`. Memory-only (#2202). */
|
||||
pdgEmitChunkSize?: number;
|
||||
/**
|
||||
* Default branch threaded into generated AGENTS.md / CLAUDE.md so the
|
||||
* regression-compare example uses the configured branch instead of a
|
||||
|
|
@ -426,6 +440,48 @@ export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] =>
|
|||
}
|
||||
: undefined;
|
||||
|
||||
/**
|
||||
* Whether streaming/chunked PDG graph emit (#2202) engages this run.
|
||||
*
|
||||
* Streaming flushes the BasicBlock + intra-file PDG-edge layer to CSV-on-disk
|
||||
* during the emit loop and never lands it in the in-memory graph, bounding peak
|
||||
* RSS to O(chunk). It is sound ONLY on a full rebuild: the incremental
|
||||
* writeback (`extractChangedSubgraph`) reads BasicBlock nodes back out of the
|
||||
* in-memory graph, which streaming has already offloaded. `force === true` is
|
||||
* the pre-pipeline guarantee of a full rebuild — `isIncremental` has
|
||||
* `!force` as a necessary condition — so gating on it avoids the deliberately
|
||||
* absent pre-pipeline incremental prediction (see the `isIncremental` note).
|
||||
*
|
||||
* Requires `pdg === true` (nothing to stream otherwise). Enabled by either the
|
||||
* explicit `streamPdgEmit` option or the `GITNEXUS_STREAM_PDG_EMIT` env toggle.
|
||||
* Memory-only — NOT part of {@link resolvePdgConfig}, so toggling it never
|
||||
* trips `pdgModeMismatch`. Read every call (not memoized) so `vi.stubEnv`
|
||||
* works in tests. Pure + exported for testing.
|
||||
*/
|
||||
export const resolveStreamPdgEmit = (options: {
|
||||
pdg?: boolean;
|
||||
force?: boolean;
|
||||
streamPdgEmit?: boolean;
|
||||
}): boolean =>
|
||||
options.pdg === true &&
|
||||
options.force === true &&
|
||||
(options.streamPdgEmit === true || parseTruthyEnv(process.env.GITNEXUS_STREAM_PDG_EMIT));
|
||||
|
||||
/**
|
||||
* Resolve the streamed PDG-emit write-buffer size (#2202). Explicit option wins
|
||||
* over `GITNEXUS_PDG_EMIT_CHUNK_SIZE`; `undefined` ⇒ the sink's
|
||||
* `DEFAULT_PDG_EMIT_CHUNK_ROWS`. Memory-only; does not affect emitted bytes.
|
||||
*/
|
||||
export const resolvePdgEmitChunkSize = (options: {
|
||||
pdgEmitChunkSize?: number;
|
||||
}): number | undefined => {
|
||||
// Only honor a positive-integer explicit option; `0`/negative is NOT nullish
|
||||
// so `?? env` would pass it through and make the sink flush every row.
|
||||
const opt = options.pdgEmitChunkSize;
|
||||
if (opt !== undefined && Number.isInteger(opt) && opt > 0) return opt;
|
||||
return parsePositiveIntEnv(process.env.GITNEXUS_PDG_EMIT_CHUNK_SIZE);
|
||||
};
|
||||
|
||||
/**
|
||||
* Whether the requested `--pdg` configuration differs from the one the
|
||||
* existing index's DB rows were built under (#2099 F1). An absent recorded
|
||||
|
|
@ -828,6 +884,11 @@ export async function runFullAnalysis(
|
|||
pdgMaxInterprocFindings: options.pdgMaxInterprocFindings,
|
||||
pdgMaxInterprocHops: options.pdgMaxInterprocHops,
|
||||
pdgMaxInterprocEdges: options.pdgMaxInterprocEdges,
|
||||
// Streaming/chunked PDG emit (#2202) — gated to full-rebuild runs
|
||||
// (force === true) so the incremental writeback never reads back an
|
||||
// offloaded BasicBlock layer. Memory-only; byte-identical output.
|
||||
streamPdgEmit: resolveStreamPdgEmit(options),
|
||||
pdgEmitChunkSize: resolvePdgEmitChunkSize(options),
|
||||
fetchWrappers: options.fetchWrappers,
|
||||
},
|
||||
);
|
||||
|
|
@ -1069,11 +1130,21 @@ export async function runFullAnalysis(
|
|||
});
|
||||
} else {
|
||||
// ── Full rebuild ───────────────────────────────────────────────
|
||||
await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => {
|
||||
lbugMsgCount++;
|
||||
const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24));
|
||||
progress('lbug', pct, msg);
|
||||
});
|
||||
// Pass the streamed PDG-emit manifest (#2202) so the BasicBlock layer that
|
||||
// was flushed to CSV during the emit loop is COPY'd alongside the
|
||||
// structural CSVs. Only ever set on a full rebuild (streaming is
|
||||
// force-gated), so the incremental branch above never carries it.
|
||||
await loadGraphToLbug(
|
||||
pipelineResult.graph,
|
||||
pipelineResult.repoPath,
|
||||
storagePath,
|
||||
(msg) => {
|
||||
lbugMsgCount++;
|
||||
const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24));
|
||||
progress('lbug', pct, msg);
|
||||
},
|
||||
pipelineResult.pdgEmitManifest,
|
||||
);
|
||||
}
|
||||
|
||||
// ── Phase 3: FTS (85–90%) ─────────────────────────────────────────
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ import type { KnowledgeGraph } from '../core/graph/types.js';
|
|||
import { CommunityDetectionResult } from '../core/ingestion/community-processor.js';
|
||||
import { ProcessDetectionResult } from '../core/ingestion/process-processor.js';
|
||||
import type { ResolutionOutcome } from '../core/ingestion/scope-resolution/resolution-outcome.js';
|
||||
import type { PdgEmitManifest } from '../core/lbug/pdg-emit-sink.js';
|
||||
|
||||
// CLI-specific: in-memory result with graph + detection results
|
||||
export interface PipelineResult {
|
||||
|
|
@ -27,4 +28,12 @@ export interface PipelineResult {
|
|||
* affordance so regression suites can prove the pool engaged.
|
||||
*/
|
||||
usedWorkerPool: boolean;
|
||||
/**
|
||||
* Streamed PDG-emit COPY manifest (#2202). Present only when streaming/chunked
|
||||
* PDG emit was active (full rebuild + `--pdg` + enabled): the BasicBlock node
|
||||
* CSV + per-pair PDG-edge CSVs flushed to disk during the emit loop, for
|
||||
* `loadGraphToLbug` to COPY alongside the structural CSVs. Absent ⇒ the PDG
|
||||
* layer (if any) is resident in `graph` and persists via the whole-graph emit.
|
||||
*/
|
||||
pdgEmitManifest?: PdgEmitManifest;
|
||||
}
|
||||
|
|
|
|||
28
gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue
Normal file
28
gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
<!--
|
||||
Vue SFC that imports a sibling TypeScript module (`./shared`). The Vue
|
||||
provider's `collectScopeContextPaths` follows that import and adds shared.ts
|
||||
to the Vue resolution pass, so shared.ts is PDG-emitted in BOTH the
|
||||
TypeScript pass and the Vue context pass — the cross-pass double-emit the
|
||||
#2202 streaming per-file dedup must collapse (review #8a). The <script>'s own
|
||||
functions add a second, Vue-side CFG so the graph holds blocks from both
|
||||
files.
|
||||
-->
|
||||
<template>
|
||||
<div class="panel" @click="onClick">{{ label }}</div>
|
||||
</template>
|
||||
|
||||
<script setup lang="ts">
|
||||
import { ref } from 'vue';
|
||||
import { classify, accumulate, guard } from './shared';
|
||||
|
||||
const count = ref(0);
|
||||
const label = classify(guard(accumulate(10)));
|
||||
|
||||
function onClick(): void {
|
||||
if (count.value > 0) {
|
||||
count.value = count.value - 1;
|
||||
} else {
|
||||
count.value = 0;
|
||||
}
|
||||
}
|
||||
</script>
|
||||
44
gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts
Normal file
44
gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts
Normal file
|
|
@ -0,0 +1,44 @@
|
|||
// Shared TypeScript module imported by app.vue. Because app.vue's
|
||||
// `collectScopeContextPaths` does a transitive import closure, THIS file is
|
||||
// pulled into the Vue scope-resolution pass IN ADDITION to the primary
|
||||
// TypeScript pass — so its worker-built CFG (the functions below) is
|
||||
// PDG-emitted in BOTH passes over the same `cfgSideChannel`, producing
|
||||
// identical BasicBlock + PDG-edge ids. The in-memory graph dedups those by id
|
||||
// (first-writer-wins Map); the streaming sink relies on per-file dedup in
|
||||
// run.ts (`pdgEmittedFiles`). This is the real cross-pass double-emit the
|
||||
// #2202 streaming dedup must collapse (review #8a).
|
||||
|
||||
export function classify(x: number): string {
|
||||
let label: string;
|
||||
if (x > 0) {
|
||||
label = 'positive';
|
||||
} else if (x < 0) {
|
||||
label = 'negative';
|
||||
} else {
|
||||
label = 'zero';
|
||||
}
|
||||
return label;
|
||||
}
|
||||
|
||||
export function accumulate(n: number): number {
|
||||
let sum = 0;
|
||||
for (let i = 0; i < n; i++) {
|
||||
if (i % 2 === 0) {
|
||||
sum += i;
|
||||
} else {
|
||||
sum -= 1;
|
||||
}
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
export function guard(value: number): number {
|
||||
if (value > 100) {
|
||||
return clamp(value);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
function clamp(value: number): number {
|
||||
return value > 100 ? 100 : value;
|
||||
}
|
||||
167
gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts
Normal file
167
gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts
Normal file
|
|
@ -0,0 +1,167 @@
|
|||
/**
|
||||
* End-to-end proof of streaming/chunked PDG graph emit (issue #2202).
|
||||
*
|
||||
* Runs the real pipeline (workers + scope-resolution) on the pdg-repo fixture
|
||||
* TWICE — a non-streamed baseline and a streamed run — and asserts:
|
||||
* - the streamed run's in-memory graph holds ZERO BasicBlock nodes and ZERO
|
||||
* intra-file PDG edges (the bulky layer was flushed to CSV, never resident
|
||||
* — the O(chunk) RSS bound, R1);
|
||||
* - the streamed run produces a `pdgEmitManifest` whose BasicBlock + PDG-edge
|
||||
* row counts EQUAL the baseline's resident counts (same emitted SET, R2);
|
||||
* - the baseline (streaming off) still emits the PDG layer into the graph,
|
||||
* i.e. the default path is unchanged (R3).
|
||||
*
|
||||
* Both runs use a durable parse cache (streaming requires `parseCache.storagePath`
|
||||
* for its CSV dir), differing ONLY in `streamPdgEmit`, so streaming is the only
|
||||
* variable.
|
||||
*/
|
||||
import { describe, it, expect, afterAll } from 'vitest';
|
||||
import fs from 'fs';
|
||||
import os from 'os';
|
||||
import path from 'path';
|
||||
import { runPipelineFromRepo } from '../../../src/core/ingestion/pipeline.js';
|
||||
import { loadParseCache } from '../../../src/storage/parse-cache.js';
|
||||
import type { PipelineResult } from '../../../src/types/pipeline.js';
|
||||
|
||||
const FIXTURE = path.join(__dirname, 'fixtures', 'pdg-repo');
|
||||
// A `.vue` SFC importing a `.ts` module: the TS module is PDG-emitted in BOTH
|
||||
// the TypeScript pass and the Vue context pass (review #8a, cross-pass dedup).
|
||||
const VUE_TS_FIXTURE = path.join(__dirname, 'fixtures', 'vue-ts-pdg');
|
||||
const PDG_EDGE_TYPES = new Set([
|
||||
'CFG',
|
||||
'REACHING_DEF',
|
||||
'CDG',
|
||||
'POST_DOMINATE',
|
||||
'TAINTED',
|
||||
'SANITIZES',
|
||||
]);
|
||||
|
||||
const tmpDirs: string[] = [];
|
||||
function freshRepo(fixture: string = FIXTURE): string {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-pdg-stream-'));
|
||||
fs.cpSync(fixture, dir, { recursive: true });
|
||||
tmpDirs.push(dir);
|
||||
return dir;
|
||||
}
|
||||
function freshStorage(): string {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-pdg-store-'));
|
||||
tmpDirs.push(dir);
|
||||
return dir;
|
||||
}
|
||||
|
||||
function pdgCounts(result: PipelineResult): { basicBlocks: number; pdgEdges: number } {
|
||||
let basicBlocks = 0;
|
||||
result.graph.forEachNode((n) => {
|
||||
if (n.label === 'BasicBlock') basicBlocks++;
|
||||
});
|
||||
let pdgEdges = 0;
|
||||
for (const rel of result.graph.iterRelationships()) {
|
||||
if (PDG_EDGE_TYPES.has(rel.type)) pdgEdges++;
|
||||
}
|
||||
return { basicBlocks, pdgEdges };
|
||||
}
|
||||
|
||||
describe('#2202 — streaming PDG emit end-to-end', () => {
|
||||
afterAll(() => {
|
||||
for (const d of tmpDirs) fs.rmSync(d, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it('streams the PDG layer out of the graph while preserving the emitted set', async () => {
|
||||
// ── Baseline: --pdg on, streaming OFF (durable cache, same as streamed) ──
|
||||
const baseStorage = freshStorage();
|
||||
const baseline = await runPipelineFromRepo(freshRepo(), () => {}, {
|
||||
pdg: true,
|
||||
parseCache: await loadParseCache(baseStorage),
|
||||
});
|
||||
const base = pdgCounts(baseline);
|
||||
// R3 / sanity: the default path still materializes the PDG layer in-graph.
|
||||
expect(base.basicBlocks).toBeGreaterThan(0);
|
||||
expect(base.pdgEdges).toBeGreaterThan(0);
|
||||
expect(baseline.pdgEmitManifest).toBeUndefined();
|
||||
|
||||
// ── Streamed: --pdg on, streaming ON ─────────────────────────────────
|
||||
const streamStorage = freshStorage();
|
||||
const streamed = await runPipelineFromRepo(freshRepo(), () => {}, {
|
||||
pdg: true,
|
||||
streamPdgEmit: true,
|
||||
parseCache: await loadParseCache(streamStorage),
|
||||
});
|
||||
const streamedCounts = pdgCounts(streamed);
|
||||
|
||||
// R1: the bulky PDG layer never accumulated in the in-memory graph.
|
||||
expect(streamedCounts.basicBlocks).toBe(0);
|
||||
expect(streamedCounts.pdgEdges).toBe(0);
|
||||
|
||||
// R2: the streamed manifest carries the SAME emitted set as the baseline.
|
||||
const manifest = streamed.pdgEmitManifest;
|
||||
expect(manifest).toBeDefined();
|
||||
const bbRows = manifest!.nodeFiles.get('BasicBlock')?.rows ?? 0;
|
||||
expect(bbRows).toBe(base.basicBlocks);
|
||||
let manifestEdgeRows = 0;
|
||||
for (const [, meta] of manifest!.relsByPair) manifestEdgeRows += meta.rows;
|
||||
expect(manifestEdgeRows).toBe(base.pdgEdges);
|
||||
|
||||
// The streamed BasicBlock CSV exists on disk under the storage dir.
|
||||
const bbCsv = manifest!.nodeFiles.get('BasicBlock')?.csvPath;
|
||||
expect(bbCsv).toBeDefined();
|
||||
expect(fs.existsSync(bbCsv!)).toBe(true);
|
||||
});
|
||||
|
||||
it('collapses the real Vue+TS cross-pass double-emit to one streamed copy (review #8a)', async () => {
|
||||
// A `.ts` module imported by a `.vue` SFC is PDG-emitted in BOTH the
|
||||
// TypeScript pass and the Vue context pass (the Vue provider's
|
||||
// `collectScopeContextPaths` follows the import) over the same worker-built
|
||||
// `cfgSideChannel` → identical ids. The in-memory graph dedups those by id
|
||||
// (Map first-writer-wins); the streaming sink is dedup-free, so the emit
|
||||
// loop dedups per FILE via `pdgEmittedFiles`. Without that dedup the streamed
|
||||
// manifest would carry shared.ts's blocks TWICE (verified out-of-band: 33 →
|
||||
// 61 BasicBlock rows). This asserts the streamed SET equals the Map-deduped
|
||||
// baseline — the load-bearing dedup regression guard.
|
||||
const baseStorage = freshStorage();
|
||||
const baseline = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, {
|
||||
pdg: true,
|
||||
parseCache: await loadParseCache(baseStorage),
|
||||
});
|
||||
const base = pdgCounts(baseline);
|
||||
// Both files contribute blocks — the cross-pass case is actually present.
|
||||
expect(base.basicBlocks).toBeGreaterThan(0);
|
||||
expect(base.pdgEdges).toBeGreaterThan(0);
|
||||
|
||||
const streamStorage = freshStorage();
|
||||
const streamed = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, {
|
||||
pdg: true,
|
||||
streamPdgEmit: true,
|
||||
parseCache: await loadParseCache(streamStorage),
|
||||
});
|
||||
// Streamed graph holds none of the PDG layer.
|
||||
const streamedCounts = pdgCounts(streamed);
|
||||
expect(streamedCounts.basicBlocks).toBe(0);
|
||||
expect(streamedCounts.pdgEdges).toBe(0);
|
||||
|
||||
// The streamed manifest carries each file's PDG layer EXACTLY ONCE — equal
|
||||
// to the Map-deduped baseline. A broken per-file dedup would double the
|
||||
// shared module's rows and fail here.
|
||||
const manifest = streamed.pdgEmitManifest;
|
||||
expect(manifest).toBeDefined();
|
||||
expect(manifest!.nodeFiles.get('BasicBlock')?.rows ?? 0).toBe(base.basicBlocks);
|
||||
let manifestEdgeRows = 0;
|
||||
for (const [, meta] of manifest!.relsByPair) manifestEdgeRows += meta.rows;
|
||||
expect(manifestEdgeRows).toBe(base.pdgEdges);
|
||||
});
|
||||
|
||||
it('falls back to in-memory emit when streaming is on but no storage path exists', async () => {
|
||||
// `streamPdgEmit: true` with NO parse cache → `parsedFileStorePath` is
|
||||
// undefined, so phase.ts cannot place the streamed CSV dir and falls back to
|
||||
// the in-memory whole-graph emit (the `else` branch that warns). The PDG
|
||||
// layer must still land in the graph and NO manifest is produced.
|
||||
const fellBack = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, {
|
||||
pdg: true,
|
||||
streamPdgEmit: true,
|
||||
// intentionally no parseCache → no storagePath
|
||||
});
|
||||
const counts = pdgCounts(fellBack);
|
||||
expect(counts.basicBlocks).toBeGreaterThan(0); // emitted in-memory, not streamed
|
||||
expect(counts.pdgEdges).toBeGreaterThan(0);
|
||||
expect(fellBack.pdgEmitManifest).toBeUndefined(); // no streaming happened
|
||||
});
|
||||
});
|
||||
|
|
@ -4,17 +4,45 @@
|
|||
* Tests: streamAllCSVsToDisk with real graph data.
|
||||
* Covers hardening fixes: LRU cache (#24), BufferedCSVWriter flush
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
|
||||
import { describe, it, expect, beforeAll, beforeEach, afterAll } from 'vitest';
|
||||
import fs from 'fs/promises';
|
||||
import { finished } from 'stream/promises';
|
||||
import path from 'path';
|
||||
import { createTempDir, type TestDBHandle } from '../helpers/test-db.js';
|
||||
import { buildTestGraph, type TestNodeInput, type TestRelInput } from '../helpers/test-graph.js';
|
||||
import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.js';
|
||||
import {
|
||||
streamAllCSVsToDisk,
|
||||
buildRelRow,
|
||||
REL_CSV_HEADER,
|
||||
} from '../../src/core/lbug/csv-generator.js';
|
||||
import { splitRelCsvByLabelPair } from '../../src/core/lbug/lbug-adapter.js';
|
||||
import { getNodeLabel } from '../../src/core/lbug/rel-pair-routing.js';
|
||||
import { NODE_TABLES } from '../../src/core/lbug/schema.js';
|
||||
|
||||
let tmpHandle: TestDBHandle;
|
||||
let csvDir: string;
|
||||
let repoDir: string;
|
||||
|
||||
/** Data rows (header dropped) of one CSV file's text. */
|
||||
const dataRowsOf = (csv: string): string[] =>
|
||||
csv
|
||||
.trim()
|
||||
.split('\n')
|
||||
.slice(1)
|
||||
.filter((l) => l.length > 0);
|
||||
|
||||
/** Concatenate data rows from every per-pair rel file (#2203 U2), pair keys
|
||||
* sorted so the concatenation order is deterministic regardless of map order. */
|
||||
const readAllRelRows = async (
|
||||
relsByPair: Map<string, { csvPath: string; rows: number }>,
|
||||
): Promise<string[]> => {
|
||||
const rows: string[] = [];
|
||||
for (const key of [...relsByPair.keys()].sort()) {
|
||||
rows.push(...dataRowsOf(await fs.readFile(relsByPair.get(key)!.csvPath, 'utf-8')));
|
||||
}
|
||||
return rows;
|
||||
};
|
||||
|
||||
beforeAll(async () => {
|
||||
tmpHandle = await createTempDir('csv-pipeline-test-');
|
||||
csvDir = path.join(tmpHandle.dbPath, 'csv');
|
||||
|
|
@ -76,9 +104,9 @@ describe('streamAllCSVsToDisk', () => {
|
|||
{ id: 'folder:src', label: 'Folder', name: 'src', filePath: 'src' },
|
||||
],
|
||||
[
|
||||
{ sourceId: 'func:main', targetId: 'func:helper', type: 'CALLS' },
|
||||
{ sourceId: 'file:src/index.ts', targetId: 'func:main', type: 'CONTAINS' },
|
||||
{ sourceId: 'file:src/utils.ts', targetId: 'func:helper', type: 'CONTAINS' },
|
||||
{ sourceId: 'Function:main', targetId: 'Function:helper', type: 'CALLS' },
|
||||
{ sourceId: 'File:src/index.ts', targetId: 'Function:main', type: 'CONTAINS' },
|
||||
{ sourceId: 'File:src/utils.ts', targetId: 'Function:helper', type: 'CONTAINS' },
|
||||
],
|
||||
);
|
||||
|
||||
|
|
@ -86,7 +114,8 @@ describe('streamAllCSVsToDisk', () => {
|
|||
|
||||
// Check that CSV files were created
|
||||
expect(result.nodeFiles.size).toBeGreaterThan(0);
|
||||
expect(result.relRows).toBe(3);
|
||||
expect(result.totalValidRels).toBe(3);
|
||||
expect(result.skippedRels).toBe(0);
|
||||
|
||||
// Verify File CSV
|
||||
const fileCsv = result.nodeFiles.get('File');
|
||||
|
|
@ -108,10 +137,12 @@ describe('streamAllCSVsToDisk', () => {
|
|||
expect(folderCsv).toBeDefined();
|
||||
expect(folderCsv!.rows).toBe(1);
|
||||
|
||||
// Verify relations CSV exists
|
||||
const relContent = await fs.readFile(result.relCsvPath, 'utf-8');
|
||||
const relLines = relContent.trim().split('\n');
|
||||
expect(relLines.length).toBe(4); // header + 3 relationships
|
||||
// Relationships are routed to per-FROM→TO-label-pair files (#2203 U2):
|
||||
// Function→Function (CALLS) + File→Function (2× CONTAINS).
|
||||
expect(result.relsByPair.has('Function|Function')).toBe(true);
|
||||
expect(result.relsByPair.has('File|Function')).toBe(true);
|
||||
expect(result.relsByPair.get('File|Function')!.rows).toBe(2);
|
||||
expect(await readAllRelRows(result.relsByPair)).toHaveLength(3);
|
||||
});
|
||||
|
||||
it('CSV content is properly escaped', async () => {
|
||||
|
|
@ -210,7 +241,8 @@ describe('streamAllCSVsToDisk', () => {
|
|||
const graph = buildTestGraph([], []);
|
||||
const result = await streamAllCSVsToDisk(graph, repoDir, csvDir);
|
||||
expect(result.nodeFiles.size).toBe(0);
|
||||
expect(result.relRows).toBe(0);
|
||||
expect(result.totalValidRels).toBe(0);
|
||||
expect(result.relsByPair.size).toBe(0);
|
||||
});
|
||||
|
||||
it('handles node with empty string properties', async () => {
|
||||
|
|
@ -221,6 +253,27 @@ describe('streamAllCSVsToDisk', () => {
|
|||
expect(fileCsv).toBeDefined();
|
||||
expect(fileCsv!.rows).toBe(1);
|
||||
});
|
||||
|
||||
it('crosses the BufferedCSVWriter FLUSH_EVERY boundary without losing rows', async () => {
|
||||
// FLUSH_EVERY=500; a >500-node graph forces ≥1 mid-stream flush, exercising
|
||||
// addRow's flush-promise return + the loop's `if (pending) await pending`
|
||||
// path that the small fixtures above never reach (only the bench did).
|
||||
const N = 600;
|
||||
const nodes = Array.from({ length: N }, (_, i) => ({
|
||||
id: `File:src/f${i}.ts`,
|
||||
label: 'File' as const,
|
||||
name: `f${i}.ts`,
|
||||
filePath: `src/f${i}.ts`,
|
||||
}));
|
||||
const result = await streamAllCSVsToDisk(buildTestGraph(nodes), repoDir, csvDir);
|
||||
|
||||
const fileCsv = result.nodeFiles.get('File');
|
||||
expect(fileCsv).toBeDefined();
|
||||
expect(fileCsv!.rows).toBe(N); // no rows dropped/duplicated at the flush boundary
|
||||
const dataRows = dataRowsOf(await fs.readFile(fileCsv!.csvPath, 'utf-8'));
|
||||
expect(dataRows).toHaveLength(N);
|
||||
expect(new Set(dataRows).size).toBe(N); // all distinct — no flush-boundary corruption
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
|
|
@ -234,15 +287,18 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => {
|
|||
// Folder nodes: single-line CSV rows (no multi-line `content` column), so the
|
||||
// id is the first comma-separated field and split('\n') is safe. ids are
|
||||
// deliberately NOT in insertion order (c, a, b).
|
||||
// ids use the `Folder:` prefix so getNodeLabel derives the valid `Folder`
|
||||
// table — edges route to rel_Folder_Folder.csv (#2203 U2). Deliberately NOT
|
||||
// in insertion order (c, a, b).
|
||||
const NODES: TestNodeInput[] = [
|
||||
{ id: 'folder:c', label: 'Folder', name: 'c', filePath: 'c' },
|
||||
{ id: 'folder:a', label: 'Folder', name: 'a', filePath: 'a' },
|
||||
{ id: 'folder:b', label: 'Folder', name: 'b', filePath: 'b' },
|
||||
{ id: 'Folder:c', label: 'Folder', name: 'c', filePath: 'c' },
|
||||
{ id: 'Folder:a', label: 'Folder', name: 'a', filePath: 'a' },
|
||||
{ id: 'Folder:b', label: 'Folder', name: 'b', filePath: 'b' },
|
||||
];
|
||||
const RELS: TestRelInput[] = [
|
||||
{ sourceId: 'folder:c', targetId: 'folder:a', type: 'CONTAINS' },
|
||||
{ sourceId: 'folder:a', targetId: 'folder:b', type: 'CONTAINS' },
|
||||
{ sourceId: 'folder:b', targetId: 'folder:c', type: 'CONTAINS' },
|
||||
{ sourceId: 'Folder:c', targetId: 'Folder:a', type: 'CONTAINS' },
|
||||
{ sourceId: 'Folder:a', targetId: 'Folder:b', type: 'CONTAINS' },
|
||||
{ sourceId: 'Folder:b', targetId: 'Folder:c', type: 'CONTAINS' },
|
||||
];
|
||||
const dataRows = (csv: string): string[] =>
|
||||
csv
|
||||
|
|
@ -270,7 +326,7 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => {
|
|||
const folderIds = folderCsv
|
||||
? dataRows(await fs.readFile(folderCsv.csvPath, 'utf-8')).map(firstCol)
|
||||
: [];
|
||||
const relRows = dataRows(await fs.readFile(result.relCsvPath, 'utf-8'));
|
||||
const relRows = await readAllRelRows(result.relsByPair);
|
||||
return { folderIds, relRows };
|
||||
} finally {
|
||||
delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT;
|
||||
|
|
@ -307,3 +363,205 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => {
|
|||
expect([...onFwd.relRows].sort()).toEqual([...offFwd.relRows].sort());
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* #2203 U2 byte-identity: for all quote-free ids the direct per-pair emit must
|
||||
* produce per-pair files byte-for-byte identical to the legacy
|
||||
* splitRelCsvByLabelPair oracle run over an equivalent monolithic relations.csv
|
||||
* from the same graph. This is the load-bearing guard for "byte-identical graph
|
||||
* content" (issue acceptance). The ONE intentional divergence — ids containing a
|
||||
* double-quote, where the router (raw-id label) is more correct than the oracle
|
||||
* (regex over the escaped row) — is asserted explicitly in its own test below.
|
||||
*/
|
||||
describe('streamAllCSVsToDisk — direct per-pair emit matches the split oracle', () => {
|
||||
// The oracle always emits in graph.iterRelationships() (unsorted) order; the
|
||||
// production path honours GITNEXUS_SORT_GRAPH_OUTPUT. Clear it so a value
|
||||
// leaked from a prior test can't desync the two and produce a spurious diff.
|
||||
beforeEach(() => {
|
||||
delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT;
|
||||
});
|
||||
|
||||
it('produces byte-identical per-pair files + identical skip/total accounting', async () => {
|
||||
// Multiple valid pairs, getNodeLabel special prefixes (comm_ AND proc_), and
|
||||
// one invalid-label edge that BOTH paths must skip identically.
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:a.ts', label: 'File', name: 'a.ts', filePath: 'a.ts' },
|
||||
{ id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' },
|
||||
{ id: 'Function:a.ts:g:5', label: 'Function', name: 'g', filePath: 'a.ts' },
|
||||
{ id: 'comm_1', label: 'Community' as never, name: 'c1', filePath: '' },
|
||||
{ id: 'comm_2', label: 'Community' as never, name: 'c2', filePath: '' },
|
||||
{ id: 'proc_1', label: 'Process' as never, name: 'p1', filePath: '' },
|
||||
{ id: 'proc_2', label: 'Process' as never, name: 'p2', filePath: '' },
|
||||
],
|
||||
[
|
||||
{ sourceId: 'File:a.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' },
|
||||
{ sourceId: 'File:a.ts', targetId: 'Function:a.ts:g:5', type: 'CONTAINS' },
|
||||
{ sourceId: 'Function:a.ts:f:1', targetId: 'Function:a.ts:g:5', type: 'CALLS' },
|
||||
{ sourceId: 'comm_1', targetId: 'comm_2', type: 'CONTAINS' },
|
||||
// proc_ prefix → Process label (getNodeLabel special case).
|
||||
{ sourceId: 'proc_1', targetId: 'proc_2', type: 'CONTAINS' },
|
||||
// Invalid FROM label ('Bogus' ∉ NODE_TABLES) — skipped by both paths.
|
||||
{ sourceId: 'Bogus:x', targetId: 'File:a.ts', type: 'CONTAINS' },
|
||||
// Invalid TO label — exercises the OTHER branch of the skip condition.
|
||||
{ sourceId: 'File:a.ts', targetId: 'Bogus:y', type: 'CONTAINS' },
|
||||
],
|
||||
);
|
||||
|
||||
const directDir = path.join(csvDir, 'diff-direct');
|
||||
const oracleDir = path.join(csvDir, 'diff-oracle');
|
||||
await fs.mkdir(oracleDir, { recursive: true });
|
||||
|
||||
// Direct emit (production path).
|
||||
const direct = await streamAllCSVsToDisk(graph, repoDir, directDir);
|
||||
|
||||
// Oracle: build the monolithic relations.csv this graph would have produced
|
||||
// (same insertion order, same row bytes via buildRelRow), then split it.
|
||||
const relCsv = path.join(oracleDir, 'relations.csv');
|
||||
const lines = [REL_CSV_HEADER];
|
||||
for (const rel of graph.iterRelationships()) lines.push(buildRelRow(rel));
|
||||
await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8');
|
||||
|
||||
const split = await splitRelCsvByLabelPair(
|
||||
relCsv,
|
||||
oracleDir,
|
||||
new Set<string>(NODE_TABLES),
|
||||
getNodeLabel,
|
||||
);
|
||||
await Promise.all(
|
||||
Array.from(split.pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
|
||||
// Identical accounting.
|
||||
expect(direct.totalValidRels).toBe(split.totalValidRels);
|
||||
expect(direct.totalValidRels).toBe(5);
|
||||
expect(direct.skippedRels).toBe(split.skippedRels);
|
||||
expect(direct.skippedRels).toBe(2); // invalid-FROM + invalid-TO, both skipped
|
||||
expect(direct.relHeader).toBe(split.relHeader);
|
||||
|
||||
// Identical pair set.
|
||||
expect([...direct.relsByPair.keys()].sort()).toEqual([...split.relsByPairMeta.keys()].sort());
|
||||
|
||||
// Byte-identical per-pair file contents.
|
||||
for (const key of direct.relsByPair.keys()) {
|
||||
const directContent = await fs.readFile(direct.relsByPair.get(key)!.csvPath, 'utf-8');
|
||||
const oracleContent = await fs.readFile(split.relsByPairMeta.get(key)!.csvPath, 'utf-8');
|
||||
expect(directContent, `pair ${key}`).toBe(oracleContent);
|
||||
}
|
||||
});
|
||||
|
||||
it('quote-in-id edge: router routes it (raw-id label) while the oracle drops it — intended divergence', async () => {
|
||||
// A node id with an embedded double-quote (legal in a POSIX filePath). The
|
||||
// router derives the label from the RAW id (`File`), so it routes the edge;
|
||||
// the oracle re-derives the label via /"([^"]*)","([^"]*)"/ over the ESCAPED
|
||||
// row (`"File:a""b.ts",...`), mis-reads the field, and drops it. This locks
|
||||
// the intended divergence so a future change can't silently revert the
|
||||
// router to the buggy regex semantics.
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:clean.ts', label: 'File', name: 'clean.ts', filePath: 'clean.ts' },
|
||||
{ id: 'File:a"b.ts', label: 'File', name: 'a"b.ts', filePath: 'a"b.ts' },
|
||||
{ id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' },
|
||||
],
|
||||
[
|
||||
{ sourceId: 'File:clean.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' },
|
||||
{ sourceId: 'File:a"b.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' },
|
||||
],
|
||||
);
|
||||
|
||||
const directDir = path.join(csvDir, 'qd-direct');
|
||||
const oracleDir = path.join(csvDir, 'qd-oracle');
|
||||
await fs.mkdir(oracleDir, { recursive: true });
|
||||
|
||||
const direct = await streamAllCSVsToDisk(graph, repoDir, directDir);
|
||||
|
||||
const relCsv = path.join(oracleDir, 'relations.csv');
|
||||
const lines = [REL_CSV_HEADER];
|
||||
for (const rel of graph.iterRelationships()) lines.push(buildRelRow(rel));
|
||||
await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8');
|
||||
const split = await splitRelCsvByLabelPair(
|
||||
relCsv,
|
||||
oracleDir,
|
||||
new Set<string>(NODE_TABLES),
|
||||
getNodeLabel,
|
||||
);
|
||||
await Promise.all(
|
||||
Array.from(split.pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
|
||||
// Router routes BOTH edges — the raw-id label `File` is valid for both.
|
||||
expect(direct.totalValidRels).toBe(2);
|
||||
expect(direct.skippedRels).toBe(0);
|
||||
expect(direct.relsByPair.get('File|Function')!.rows).toBe(2);
|
||||
|
||||
// Oracle DIVERGES: its regex mis-reads the quote-in-id row and drops that
|
||||
// edge, so it routes strictly fewer edges. Asserted robustly — we do NOT
|
||||
// pin the oracle's exact mis-derived label.
|
||||
expect(split.totalValidRels).toBeLessThan(direct.totalValidRels);
|
||||
expect(split.skippedRels).toBeGreaterThan(direct.skippedRels);
|
||||
});
|
||||
|
||||
it('sorted path (GITNEXUS_SORT_GRAPH_OUTPUT=1): per-pair files byte-identical to the oracle', async () => {
|
||||
// The earlier differential test covers the default (insertion-order) path.
|
||||
// Here the sorted emit path must also match the oracle — fed the SAME
|
||||
// id-sorted order orderedRelationships() uses (sort by rel.id).
|
||||
process.env.GITNEXUS_SORT_GRAPH_OUTPUT = '1';
|
||||
try {
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:a.ts', label: 'File', name: 'a.ts', filePath: 'a.ts' },
|
||||
{ id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' },
|
||||
{ id: 'Function:a.ts:g:5', label: 'Function', name: 'g', filePath: 'a.ts' },
|
||||
],
|
||||
// Deliberately NOT in id-sorted order so the sort actually reorders rows.
|
||||
[
|
||||
{ sourceId: 'Function:a.ts:f:1', targetId: 'Function:a.ts:g:5', type: 'CALLS' },
|
||||
{ sourceId: 'File:a.ts', targetId: 'Function:a.ts:g:5', type: 'CONTAINS' },
|
||||
{ sourceId: 'File:a.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' },
|
||||
],
|
||||
);
|
||||
|
||||
const directDir = path.join(csvDir, 'sorted-direct');
|
||||
const oracleDir = path.join(csvDir, 'sorted-oracle');
|
||||
await fs.mkdir(oracleDir, { recursive: true });
|
||||
|
||||
const direct = await streamAllCSVsToDisk(graph, repoDir, directDir);
|
||||
|
||||
// Oracle fed the same id-sorted order the sorted emit produces.
|
||||
const sortedRels = [...graph.iterRelationships()].sort((a, b) =>
|
||||
a.id < b.id ? -1 : a.id > b.id ? 1 : 0,
|
||||
);
|
||||
const relCsv = path.join(oracleDir, 'relations.csv');
|
||||
const lines = [REL_CSV_HEADER];
|
||||
for (const rel of sortedRels) lines.push(buildRelRow(rel));
|
||||
await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8');
|
||||
const split = await splitRelCsvByLabelPair(
|
||||
relCsv,
|
||||
oracleDir,
|
||||
new Set<string>(NODE_TABLES),
|
||||
getNodeLabel,
|
||||
);
|
||||
await Promise.all(
|
||||
Array.from(split.pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
|
||||
expect([...direct.relsByPair.keys()].sort()).toEqual([...split.relsByPairMeta.keys()].sort());
|
||||
for (const key of direct.relsByPair.keys()) {
|
||||
const directContent = await fs.readFile(direct.relsByPair.get(key)!.csvPath, 'utf-8');
|
||||
const oracleContent = await fs.readFile(split.relsByPairMeta.get(key)!.csvPath, 'utf-8');
|
||||
expect(directContent, `pair ${key} (sorted)`).toBe(oracleContent);
|
||||
}
|
||||
} finally {
|
||||
delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
|
|
|||
141
gitnexus/test/integration/lbug-load-prof.test.ts
Normal file
141
gitnexus/test/integration/lbug-load-prof.test.ts
Normal file
|
|
@ -0,0 +1,141 @@
|
|||
/**
|
||||
* Integration test: PROF_LBUG_LOAD persistence-path profiling (#2203 U1).
|
||||
*
|
||||
* loadGraphToLbug is un-timed in production today; the analyze "emit" number
|
||||
* is the scope-resolution emit bucket, not this CSV→COPY persistence path.
|
||||
* U1 adds a zero-cost-when-off per-stage breakdown gated by PROF_LBUG_LOAD=1,
|
||||
* mirroring the PROF_SCOPE_RESOLUTION pattern. These tests assert the gate:
|
||||
* - flag off → no `[lbug-load prof]` line is logged, behaviour unchanged
|
||||
* - flag on → exactly one summary line with every stage key + node/rel counts
|
||||
*
|
||||
* Needs a real LadybugDB connection (initLbug), so it lives under integration.
|
||||
* Logger assertions use `_captureLogger()` — the exported `logger` is a Proxy
|
||||
* over a lazily-built pino instance and is not directly spy-able.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, beforeEach, afterAll, afterEach } from 'vitest';
|
||||
import fs from 'fs/promises';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import { buildTestGraph } from '../helpers/test-graph.js';
|
||||
import { _captureLogger, type LoggerCapture } from '../../src/core/logger.js';
|
||||
|
||||
let tmpBase: string;
|
||||
let storagePath: string;
|
||||
let dbPath: string;
|
||||
let cap: LoggerCapture;
|
||||
|
||||
const PROF_LINE = '[lbug-load prof]';
|
||||
|
||||
const profLines = (): string[] =>
|
||||
cap
|
||||
.records()
|
||||
.map((r) => (typeof r.msg === 'string' ? r.msg : ''))
|
||||
.filter((msg) => msg.includes(PROF_LINE));
|
||||
|
||||
beforeAll(async () => {
|
||||
tmpBase = path.join(os.tmpdir(), `gitnexus-lbug-prof-${Date.now()}-${process.pid}`);
|
||||
storagePath = path.join(tmpBase, '.gitnexus');
|
||||
dbPath = path.join(storagePath, 'lbug');
|
||||
await fs.mkdir(dbPath, { recursive: true });
|
||||
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
await adapter.initLbug(dbPath);
|
||||
});
|
||||
|
||||
beforeEach(() => {
|
||||
cap = _captureLogger();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
cap.restore();
|
||||
delete process.env.PROF_LBUG_LOAD;
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
try {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
await adapter.closeLbug();
|
||||
} catch {
|
||||
/* may not have opened */
|
||||
}
|
||||
try {
|
||||
await fs.rm(tmpBase, { recursive: true, force: true });
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
});
|
||||
|
||||
describe('PROF_LBUG_LOAD persistence-path profiling (#2203 U1)', () => {
|
||||
it('does NOT log a prof summary when the flag is unset', async () => {
|
||||
delete process.env.PROF_LBUG_LOAD;
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:src/off.ts', label: 'File', name: 'off.ts', filePath: 'src/off.ts' },
|
||||
{
|
||||
id: 'Function:src/off.ts:offFn:1',
|
||||
label: 'Function',
|
||||
name: 'offFn',
|
||||
filePath: 'src/off.ts',
|
||||
startLine: 1,
|
||||
endLine: 2,
|
||||
},
|
||||
],
|
||||
[{ sourceId: 'File:src/off.ts', targetId: 'Function:src/off.ts:offFn:1', type: 'DEFINES' }],
|
||||
);
|
||||
|
||||
const result = await adapter.loadGraphToLbug(graph, tmpBase, storagePath);
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(profLines()).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('logs exactly one summary line with all stage keys + counts when the flag is set', async () => {
|
||||
process.env.PROF_LBUG_LOAD = '1';
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
|
||||
// Distinct ids from the flag-off graph so the COPY does not hit a
|
||||
// PK-dup IGNORE_ERRORS retry on the shared singleton connection.
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:src/on.ts', label: 'File', name: 'on.ts', filePath: 'src/on.ts' },
|
||||
{
|
||||
id: 'Function:src/on.ts:onFn:1',
|
||||
label: 'Function',
|
||||
name: 'onFn',
|
||||
filePath: 'src/on.ts',
|
||||
startLine: 1,
|
||||
endLine: 2,
|
||||
},
|
||||
{
|
||||
id: 'Class:src/on.ts:OnClass:5',
|
||||
label: 'Class',
|
||||
name: 'OnClass',
|
||||
filePath: 'src/on.ts',
|
||||
startLine: 5,
|
||||
endLine: 8,
|
||||
},
|
||||
],
|
||||
[
|
||||
{ sourceId: 'File:src/on.ts', targetId: 'Function:src/on.ts:onFn:1', type: 'DEFINES' },
|
||||
{ sourceId: 'File:src/on.ts', targetId: 'Class:src/on.ts:OnClass:5', type: 'DEFINES' },
|
||||
],
|
||||
);
|
||||
|
||||
const result = await adapter.loadGraphToLbug(graph, tmpBase, storagePath);
|
||||
expect(result.success).toBe(true);
|
||||
|
||||
const lines = profLines();
|
||||
expect(lines).toHaveLength(1);
|
||||
|
||||
const line = lines[0];
|
||||
// Relationships are routed to per-pair files during csv-emit (#2203 U2),
|
||||
// so there is no separate rel-split stage.
|
||||
for (const key of ['csv-emit=', 'copy-nodes=', 'copy-rels=', 'fallback=', 'total=']) {
|
||||
expect(line).toContain(key);
|
||||
}
|
||||
// 3 node rows (File, Function, Class), 2 valid rels emitted.
|
||||
expect(line).toContain('(3 nodes, 2 rels)');
|
||||
});
|
||||
});
|
||||
181
gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts
Normal file
181
gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts
Normal file
|
|
@ -0,0 +1,181 @@
|
|||
/**
|
||||
* Integration test: streamed PDG-emit manifest round-trips the bulk-COPY load
|
||||
* path (issue #2202 U5).
|
||||
*
|
||||
* Simulates the streaming case end-to-end at the persistence boundary: the
|
||||
* BasicBlock + intra-file PDG-edge layer is flushed to CSV by a real
|
||||
* `PdgEmitSink` (so the in-memory graph holds ZERO BasicBlocks, exactly as in a
|
||||
* streamed run), and `loadGraphToLbug` is handed the resulting manifest. Asserts
|
||||
* the BasicBlock nodes + every PDG edge type land in the DB via the manifest,
|
||||
* alongside the structural graph — and that there is no double-COPY.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
|
||||
import fs from 'fs/promises';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { PdgEmitSink } from '../../src/core/lbug/pdg-emit-sink.js';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
|
||||
let tmpBase: string;
|
||||
let storagePath: string;
|
||||
|
||||
const FILE_ID = 'File:src/a.ts';
|
||||
const BB = (i: number) => `BasicBlock:src/a.ts:1:0:${i}`;
|
||||
const PDG_TYPES = ['CFG', 'REACHING_DEF', 'CDG', 'POST_DOMINATE', 'TAINTED', 'SANITIZES'] as const;
|
||||
|
||||
beforeAll(async () => {
|
||||
// mkdtemp (unpredictable, unique) — not a predictable os-temp path.
|
||||
tmpBase = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-pdg-stream-rt-'));
|
||||
storagePath = path.join(tmpBase, '.gitnexus');
|
||||
await fs.mkdir(path.join(storagePath, 'lbug'), { recursive: true });
|
||||
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
await adapter.initLbug(path.join(storagePath, 'lbug'));
|
||||
|
||||
// Structural graph — NO BasicBlock nodes (they were "streamed out").
|
||||
const graph = createKnowledgeGraph();
|
||||
graph.addNode({
|
||||
id: FILE_ID,
|
||||
label: 'File',
|
||||
properties: { name: 'a.ts', filePath: 'src/a.ts' },
|
||||
});
|
||||
|
||||
// Real sink → real manifest: route 3 BasicBlocks + one edge of each PDG type.
|
||||
const sink = new PdgEmitSink(graph, path.join(storagePath, 'pdg-csv'));
|
||||
for (let i = 0; i < 3; i++) {
|
||||
const node: GraphNode = {
|
||||
id: BB(i),
|
||||
label: 'BasicBlock',
|
||||
properties: {
|
||||
name: '',
|
||||
filePath: 'src/a.ts',
|
||||
startLine: i + 1,
|
||||
endLine: i + 2,
|
||||
text: `b${i}`,
|
||||
},
|
||||
};
|
||||
sink.addNode(node);
|
||||
}
|
||||
for (const type of PDG_TYPES) {
|
||||
const rel: GraphRelationship = {
|
||||
id: `${type}:0->1`,
|
||||
sourceId: BB(0),
|
||||
targetId: BB(1),
|
||||
type,
|
||||
confidence: 1,
|
||||
reason: type === 'REACHING_DEF' ? 'x' : type === 'CDG' ? 'T' : `${type}-edge`,
|
||||
};
|
||||
sink.addRelationship(rel);
|
||||
}
|
||||
const manifest = sink.finalize();
|
||||
|
||||
// The sink offloaded the whole PDG layer — the graph has only the File node.
|
||||
expect(graph.nodeCount).toBe(1);
|
||||
|
||||
await adapter.loadGraphToLbug(graph, tmpBase, storagePath, undefined, manifest);
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
try {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
await adapter.closeLbug();
|
||||
} catch {
|
||||
/* may not have opened */
|
||||
}
|
||||
if (tmpBase) {
|
||||
for (let attempt = 0; attempt < 5; attempt++) {
|
||||
try {
|
||||
await fs.rm(tmpBase, { recursive: true, force: true });
|
||||
return;
|
||||
} catch {
|
||||
if (attempt < 4) await new Promise((r) => setTimeout(r, 200 * (attempt + 1)));
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
describe('streamed PDG manifest → bulk COPY (#2202 U5)', () => {
|
||||
it('BasicBlock nodes from the manifest land in the DB with span + text', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
const rows = await adapter.executeQuery(
|
||||
'MATCH (n:BasicBlock) RETURN n.id AS id, n.text AS text, n.startLine AS startLine ORDER BY n.id',
|
||||
);
|
||||
expect(rows).toHaveLength(3);
|
||||
expect(rows[0].id).toBe(BB(0));
|
||||
expect(rows[0].text).toBe('b0');
|
||||
expect(Number(rows[0].startLine)).toBe(1);
|
||||
});
|
||||
|
||||
it('the structural graph (File node) loaded alongside the manifest', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
const rows = await adapter.executeQuery(
|
||||
`MATCH (f:File {id: '${FILE_ID}'}) RETURN count(f) AS c`,
|
||||
);
|
||||
expect(Number(rows[0].c)).toBe(1);
|
||||
});
|
||||
|
||||
it('every PDG edge type round-trips via the manifest (no double-COPY)', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
for (const type of PDG_TYPES) {
|
||||
const rows = await adapter.executeQuery(
|
||||
`MATCH (:BasicBlock)-[r:CodeRelation {type: '${type}'}]->(:BasicBlock) RETURN count(r) AS c`,
|
||||
);
|
||||
// Exactly one — not two (double-COPY would double these).
|
||||
expect(Number(rows[0].c), `${type} should round-trip exactly once`).toBe(1);
|
||||
}
|
||||
});
|
||||
|
||||
it('REACHING_DEF carries its variable in reason (manifest path)', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
const rows = await adapter.executeQuery(
|
||||
"MATCH (a:BasicBlock)-[r:CodeRelation {type: 'REACHING_DEF', reason: 'x'}]->(b:BasicBlock) RETURN a.id AS from, b.id AS to",
|
||||
);
|
||||
expect(rows).toHaveLength(1);
|
||||
expect(rows[0].from).toBe(BB(0));
|
||||
expect(rows[0].to).toBe(BB(1));
|
||||
});
|
||||
});
|
||||
|
||||
describe('streamed PDG manifest → disjoint-key merge guard (#2202 review #3)', () => {
|
||||
// The merge in loadGraphToLbug assumes the streamed manifest and the
|
||||
// structural csvResult are disjoint: when streaming is on the in-memory graph
|
||||
// holds ZERO BasicBlocks, so streamAllCSVsToDisk emits no basicblock.csv and
|
||||
// the manifest is the sole source. A future BasicBlock-leak-into-graph would
|
||||
// make both sides carry a "BasicBlock" entry; silently overwriting one CSV
|
||||
// with the other would drop its rows. The guard fails loudly instead.
|
||||
|
||||
it('throws when the manifest collides with a structural node CSV', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
|
||||
// A graph that DOES contain a BasicBlock → streamAllCSVsToDisk emits a
|
||||
// structural basicblock.csv (the invariant-violation scenario).
|
||||
const leakyGraph = createKnowledgeGraph();
|
||||
leakyGraph.addNode({
|
||||
id: BB(0),
|
||||
label: 'BasicBlock',
|
||||
properties: { name: '', filePath: 'src/a.ts', startLine: 1, endLine: 2, text: 'leak' },
|
||||
});
|
||||
|
||||
// A manifest that ALSO declares a BasicBlock node CSV → disjoint-key clash.
|
||||
const collideBase = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-pdg-collide-'));
|
||||
const collideStorage = path.join(collideBase, '.gitnexus');
|
||||
await fs.mkdir(collideStorage, { recursive: true });
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), path.join(collideStorage, 'pdg-csv'));
|
||||
sink.addNode({
|
||||
id: BB(1),
|
||||
label: 'BasicBlock',
|
||||
properties: { name: '', filePath: 'src/a.ts', startLine: 3, endLine: 4, text: 'm' },
|
||||
});
|
||||
const manifest = sink.finalize();
|
||||
|
||||
try {
|
||||
await expect(
|
||||
adapter.loadGraphToLbug(leakyGraph, collideBase, collideStorage, undefined, manifest),
|
||||
).rejects.toThrow(/collides with a structural node CSV for "BasicBlock"/);
|
||||
} finally {
|
||||
await fs.rm(collideBase, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
});
|
||||
|
|
@ -5,7 +5,12 @@
|
|||
* 8.3 short-name form before passing them to KuzuDB's native layer.
|
||||
*/
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { toNativeSafePath, cleanupNativePathJunctions } from '../../src/core/lbug/lbug-config.js';
|
||||
import path from 'path';
|
||||
import {
|
||||
toNativeSafePath,
|
||||
cleanupNativePathJunctions,
|
||||
resolveNativeSafeStorageDir,
|
||||
} from '../../src/core/lbug/lbug-config.js';
|
||||
|
||||
describe('toNativeSafePath', () => {
|
||||
it('returns ASCII paths unchanged on any platform', () => {
|
||||
|
|
@ -61,3 +66,52 @@ describe('toNativeSafePath', () => {
|
|||
});
|
||||
}
|
||||
});
|
||||
|
||||
describe('resolveNativeSafeStorageDir (#2202)', () => {
|
||||
it('returns <storage>/<subdir> for an ASCII storage path on any platform', () => {
|
||||
const storage = path.join('repo', '.gitnexus');
|
||||
expect(resolveNativeSafeStorageDir(storage, 'csv')).toBe(path.join(storage, 'csv'));
|
||||
expect(resolveNativeSafeStorageDir(storage, 'pdg-csv')).toBe(path.join(storage, 'pdg-csv'));
|
||||
});
|
||||
|
||||
it('keeps csv and pdg-csv distinct (no collision between structural and streamed dirs)', () => {
|
||||
const storage = path.join('repo', '.gitnexus');
|
||||
expect(resolveNativeSafeStorageDir(storage, 'csv')).not.toBe(
|
||||
resolveNativeSafeStorageDir(storage, 'pdg-csv'),
|
||||
);
|
||||
});
|
||||
|
||||
if (process.platform !== 'win32') {
|
||||
it('does NOT relocate a non-ASCII storage path off Windows (platform gate)', () => {
|
||||
const storage = path.join('repo', '用户', '.gitnexus');
|
||||
// Non-win32: the relocation never fires regardless of non-ASCII chars.
|
||||
expect(resolveNativeSafeStorageDir(storage, 'pdg-csv')).toBe(path.join(storage, 'pdg-csv'));
|
||||
});
|
||||
}
|
||||
|
||||
if (process.platform === 'win32') {
|
||||
it('relocates a non-ASCII storage path to a unique mkdtemp os.tmpdir() dir per subdir', () => {
|
||||
const fs = require('fs');
|
||||
const storage = 'C:\\Project\\中文\\.gitnexus';
|
||||
// mkdtemp creates the dirs — track + clean them up.
|
||||
const csv = resolveNativeSafeStorageDir(storage, 'csv');
|
||||
const pdg = resolveNativeSafeStorageDir(storage, 'pdg-csv');
|
||||
try {
|
||||
// Both relocated under os.tmpdir(), ASCII-prefixed, and distinct (each
|
||||
// mkdtemp call returns a fresh random suffix — never a predictable name).
|
||||
expect(pdg.includes('gitnexus-pdg-csv-')).toBe(true);
|
||||
expect(csv.includes('gitnexus-csv-')).toBe(true);
|
||||
expect(csv).not.toBe(pdg);
|
||||
// Two calls for the same (storage, subdir) yield DIFFERENT dirs (random).
|
||||
const csv2 = resolveNativeSafeStorageDir(storage, 'csv');
|
||||
expect(csv2).not.toBe(csv);
|
||||
// The relocated paths are not under the original non-ASCII storage path.
|
||||
expect(pdg.includes('中文')).toBe(false);
|
||||
fs.rmSync(csv2, { recursive: true, force: true });
|
||||
} finally {
|
||||
fs.rmSync(csv, { recursive: true, force: true });
|
||||
fs.rmSync(pdg, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
|
|
|
|||
331
gitnexus/test/unit/lbug/pdg-emit-sink.test.ts
Normal file
331
gitnexus/test/unit/lbug/pdg-emit-sink.test.ts
Normal file
|
|
@ -0,0 +1,331 @@
|
|||
/**
|
||||
* PdgEmitSink unit tests (issue #2202 U2).
|
||||
*
|
||||
* Verifies the streaming PDG emit sink:
|
||||
* - routes BasicBlock nodes + PDG edges to bounded CSV-on-disk;
|
||||
* - delegates structural nodes/edges + the whole-program TAINT_PATH edge to
|
||||
* the real graph (never streamed);
|
||||
* - is byte-identical (set-wise) to the whole-graph `streamAllCSVsToDisk`
|
||||
* emit for the same node/edge set (the issue's byte-identity acceptance);
|
||||
* - never accumulates the PDG layer in the in-memory graph (the RSS bound).
|
||||
*/
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import fs from 'node:fs';
|
||||
import fsp from 'node:fs/promises';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
|
||||
import { createKnowledgeGraph } from '../../../src/core/graph/graph.js';
|
||||
import { streamAllCSVsToDisk, buildBasicBlockRow } from '../../../src/core/lbug/csv-generator.js';
|
||||
import { PdgEmitSink } from '../../../src/core/lbug/pdg-emit-sink.js';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
|
||||
const bbNode = (fp: string, idx: number, line: number): GraphNode => ({
|
||||
id: `BasicBlock:${fp}:1:0:${idx}`,
|
||||
label: 'BasicBlock',
|
||||
properties: { name: '', filePath: fp, startLine: line, endLine: line + 1, text: `blk ${idx}` },
|
||||
});
|
||||
|
||||
const pdgEdge = (
|
||||
fp: string,
|
||||
from: number,
|
||||
to: number,
|
||||
type: GraphRelationship['type'],
|
||||
reason: string,
|
||||
): GraphRelationship => ({
|
||||
id: `${type}:${fp}:${from}->${to}`,
|
||||
sourceId: `BasicBlock:${fp}:1:0:${from}`,
|
||||
targetId: `BasicBlock:${fp}:1:0:${to}`,
|
||||
type,
|
||||
confidence: 1,
|
||||
reason,
|
||||
});
|
||||
|
||||
/** Sorted non-empty lines of a CSV file (order-independent comparison). */
|
||||
const sortedLines = async (csvPath: string): Promise<string[]> => {
|
||||
const text = await fsp.readFile(csvPath, 'utf8');
|
||||
return text
|
||||
.split('\n')
|
||||
.filter((l) => l.length > 0)
|
||||
.sort();
|
||||
};
|
||||
|
||||
let tmpRoot: string;
|
||||
|
||||
beforeEach(() => {
|
||||
tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'pdg-sink-'));
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
fs.rmSync(tmpRoot, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — routing', () => {
|
||||
it('routes BasicBlock nodes and PDG edges to CSV, never to the real graph', () => {
|
||||
const real = createKnowledgeGraph();
|
||||
const sink = new PdgEmitSink(real, path.join(tmpRoot, 'pdg-csv'));
|
||||
|
||||
sink.addNode(bbNode('a.ts', 0, 1));
|
||||
sink.addNode(bbNode('a.ts', 1, 5));
|
||||
sink.addRelationship(pdgEdge('a.ts', 0, 1, 'CFG', 'seq'));
|
||||
sink.addRelationship(pdgEdge('a.ts', 0, 1, 'REACHING_DEF', 'x:1:0'));
|
||||
|
||||
// PDG layer must not land in the in-memory graph (the RSS bound).
|
||||
expect(real.nodeCount).toBe(0);
|
||||
expect(real.relationshipCount).toBe(0);
|
||||
expect(sink.nodeCount).toBe(0);
|
||||
|
||||
sink.finalize();
|
||||
});
|
||||
|
||||
it('delegates structural nodes, CALLS, and the whole-program TAINT_PATH edge to the real graph', () => {
|
||||
const real = createKnowledgeGraph();
|
||||
const sink = new PdgEmitSink(real, path.join(tmpRoot, 'pdg-csv'));
|
||||
|
||||
sink.addNode({
|
||||
id: 'Function:a.ts:fn:1',
|
||||
label: 'Function',
|
||||
properties: { name: 'fn', filePath: 'a.ts', startLine: 1, endLine: 9 },
|
||||
});
|
||||
sink.addRelationship({
|
||||
id: 'CALLS:1',
|
||||
sourceId: 'Function:a.ts:fn:1',
|
||||
targetId: 'Function:a.ts:fn2:9',
|
||||
type: 'CALLS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
// TAINT_PATH is a whole-program (Function→Function) edge — NOT streamed.
|
||||
sink.addRelationship({
|
||||
id: 'TAINT_PATH:1',
|
||||
sourceId: 'Function:a.ts:fn:1',
|
||||
targetId: 'Function:a.ts:fn2:9',
|
||||
type: 'TAINT_PATH',
|
||||
confidence: 0.9,
|
||||
reason: 'src->sink',
|
||||
});
|
||||
|
||||
expect(real.nodeCount).toBe(1);
|
||||
expect(real.relationshipCount).toBe(2);
|
||||
expect(real.getNode('Function:a.ts:fn:1')).toBeDefined();
|
||||
|
||||
const manifest = sink.finalize();
|
||||
// No BasicBlock node CSV was created (no BasicBlock nodes were routed).
|
||||
expect(manifest.nodeFiles.size).toBe(0);
|
||||
expect(manifest.relsByPair.size).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — byte-identity vs whole-graph emit', () => {
|
||||
it('streamed CSV line set equals streamAllCSVsToDisk for the same nodes/edges', async () => {
|
||||
const fp = 'a.ts';
|
||||
const nodes = [bbNode(fp, 0, 1), bbNode(fp, 1, 5), bbNode(fp, 2, 9)];
|
||||
const edges: GraphRelationship[] = [
|
||||
pdgEdge(fp, 0, 1, 'CFG', 'seq'),
|
||||
pdgEdge(fp, 1, 2, 'CFG', 'cond-true'),
|
||||
pdgEdge(fp, 0, 2, 'REACHING_DEF', 'x:1:0'),
|
||||
pdgEdge(fp, 1, 2, 'CDG', 'T'),
|
||||
pdgEdge(fp, 0, 1, 'POST_DOMINATE', ''),
|
||||
pdgEdge(fp, 0, 2, 'TAINTED', 'taint'),
|
||||
pdgEdge(fp, 1, 2, 'SANITIZES', 'clean'),
|
||||
];
|
||||
|
||||
// Whole-graph path: add to a plain graph, run streamAllCSVsToDisk.
|
||||
const wholeGraph = createKnowledgeGraph();
|
||||
for (const n of nodes) wholeGraph.addNode(n);
|
||||
for (const e of edges) wholeGraph.addRelationship(e);
|
||||
const wholeDir = path.join(tmpRoot, 'csv');
|
||||
await streamAllCSVsToDisk(wholeGraph, path.join(tmpRoot, 'no-such-repo'), wholeDir);
|
||||
|
||||
// Streamed path: route the same set through the sink.
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
for (const n of nodes) sink.addNode(n);
|
||||
for (const e of edges) sink.addRelationship(e);
|
||||
const manifest = sink.finalize();
|
||||
|
||||
// BasicBlock node CSV: identical line set.
|
||||
expect(await sortedLines(path.join(pdgDir, 'basicblock.csv'))).toEqual(
|
||||
await sortedLines(path.join(wholeDir, 'basicblock.csv')),
|
||||
);
|
||||
|
||||
// PDG edges all route to the BasicBlock|BasicBlock pair file: identical set.
|
||||
expect(await sortedLines(path.join(pdgDir, 'rel_BasicBlock_BasicBlock.csv'))).toEqual(
|
||||
await sortedLines(path.join(wholeDir, 'rel_BasicBlock_BasicBlock.csv')),
|
||||
);
|
||||
|
||||
// Manifest reports the streamed files + row counts.
|
||||
expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(nodes.length);
|
||||
expect(manifest.relsByPair.get('BasicBlock|BasicBlock')?.rows).toBe(edges.length);
|
||||
});
|
||||
|
||||
it('emits rows via the shared builder (buildBasicBlockRow)', async () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
const n = bbNode('a.ts', 0, 3);
|
||||
sink.addNode(n);
|
||||
sink.finalize();
|
||||
const lines = await sortedLines(path.join(pdgDir, 'basicblock.csv'));
|
||||
// header + one data row; the data row is exactly buildBasicBlockRow(n).
|
||||
expect(lines).toContain(buildBasicBlockRow(n));
|
||||
});
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — bounded retention', () => {
|
||||
it('flushes incrementally so the graph never holds the PDG layer', async () => {
|
||||
const real = createKnowledgeGraph();
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const CHUNK = 2;
|
||||
const sink = new PdgEmitSink(real, pdgDir, CHUNK); // tiny chunk to force flushes
|
||||
|
||||
// TOTAL is intentionally NOT a multiple of CHUNK so the final partial chunk
|
||||
// is genuinely still buffered (unflushed) at the mid-stream read. With a
|
||||
// multiple (e.g. 50 % 2 === 0) the last addRow's flush would have written
|
||||
// every row and the "mid-stream" assertion would prove nothing (#2202
|
||||
// review #7).
|
||||
const TOTAL = 51;
|
||||
const REMAINDER = TOTAL % CHUNK; // 1 — must be non-zero
|
||||
expect(REMAINDER).toBeGreaterThan(0);
|
||||
for (let i = 0; i < TOTAL; i++) sink.addNode(bbNode('a.ts', i, i));
|
||||
|
||||
// Mid-stream (before finalize): exactly the whole flushed chunks are on
|
||||
// disk; the partial last chunk (REMAINDER rows) is still buffered in memory,
|
||||
// proving the writer streams to the OS and never buffers the whole layer.
|
||||
const midText = fs.readFileSync(path.join(pdgDir, 'basicblock.csv'), 'utf8');
|
||||
const midDataRows = midText.split('\n').filter((l) => l.length > 0).length - 1; // minus header
|
||||
expect(midDataRows).toBe(TOTAL - REMAINDER); // 50 flushed, 1 still buffered
|
||||
expect(TOTAL - midDataRows).toBe(REMAINDER); // exactly the unflushed remainder
|
||||
expect(TOTAL - midDataRows).toBeLessThanOrEqual(CHUNK); // unflushed is bounded by one chunk
|
||||
|
||||
// The in-memory graph never received a single BasicBlock.
|
||||
expect(real.nodeCount).toBe(0);
|
||||
|
||||
const manifest = sink.finalize();
|
||||
expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(TOTAL);
|
||||
const finalRows = (await sortedLines(path.join(pdgDir, 'basicblock.csv'))).length - 1;
|
||||
expect(finalRows).toBe(TOTAL); // finalize flushed the buffered remainder
|
||||
});
|
||||
|
||||
it('finalize twice throws', () => {
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), path.join(tmpRoot, 'pdg-csv'));
|
||||
sink.finalize();
|
||||
expect(() => sink.finalize()).toThrow(/twice/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — pass-through contract (dedup is the caller’s)', () => {
|
||||
// The sink does NOT dedup by id (that would retain every id → O(total ids)
|
||||
// memory, undermining the O(chunk) bound). Cross-pass dedup is done upstream,
|
||||
// per file, in run.ts (a file imported by two language passes is emitted
|
||||
// once). The sink is a faithful pass-through: it writes every id it is given
|
||||
// and must not be fed duplicates. See #2202 finding #1 + the run-loop / Vue+TS
|
||||
// integration coverage for the cross-pass dedup itself.
|
||||
it('writes every BasicBlock it is given (no id dedup in the sink)', async () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
const n = bbNode('a.ts', 0, 1);
|
||||
sink.addNode(n);
|
||||
sink.addNode(n); // sink does not dedup — both rows are written
|
||||
const manifest = sink.finalize();
|
||||
expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(2);
|
||||
expect((await sortedLines(path.join(pdgDir, 'basicblock.csv'))).length - 1).toBe(2);
|
||||
});
|
||||
|
||||
it('writes every PDG edge it is given (no id dedup in the sink)', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
const e = pdgEdge('a.ts', 0, 1, 'CFG', 'seq');
|
||||
sink.addRelationship(e);
|
||||
sink.addRelationship(e);
|
||||
const manifest = sink.finalize();
|
||||
expect(manifest.relsByPair.get('BasicBlock|BasicBlock')?.rows).toBe(2);
|
||||
});
|
||||
|
||||
it('skips a PDG edge whose endpoint label is not a node table', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
// sourceId prefix "Bogus" is not in NODE_TABLES → skipped (mirrors RelPairRouter).
|
||||
sink.addRelationship({
|
||||
id: 'CFG:bogus',
|
||||
sourceId: 'Bogus:a.ts:0',
|
||||
targetId: 'BasicBlock:a.ts:1:0:1',
|
||||
type: 'CFG',
|
||||
confidence: 1,
|
||||
reason: 'seq',
|
||||
});
|
||||
const manifest = sink.finalize();
|
||||
expect(manifest.relsByPair.size).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — IO failure poisoning (#2202 review #4/#6)', () => {
|
||||
// A streamed-write failure is an IO fault, not the CFG-logic error the emit
|
||||
// loop's per-file try/catch is built to swallow. The sink poisons the failing
|
||||
// writer (or records an open failure) so finalize fails loudly instead of
|
||||
// returning a truncated manifest that the bulk COPY would silently load.
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it('poisons the writer on a mid-stream write failure (disk-full) → finalize throws', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1); // flush every row
|
||||
|
||||
// The writer opens fine (real openSync); the flush's writeSync fails.
|
||||
const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => {
|
||||
throw new Error('ENOSPC: no space left on device');
|
||||
});
|
||||
// The throw propagates to the immediate caller (the emit loop, which would
|
||||
// swallow it as a per-file CFG error) — that is exactly why finalize must
|
||||
// re-check poison below.
|
||||
expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/ENOSPC/);
|
||||
spy.mockRestore();
|
||||
|
||||
expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/);
|
||||
});
|
||||
|
||||
it('records an openSync failure (EMFILE) → finalize throws even if the caller swallowed it', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv'); // ctor mkdir/rm run before the spy
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
|
||||
const spy = vi.spyOn(fs, 'openSync').mockImplementation(() => {
|
||||
throw new Error('EMFILE: too many open files');
|
||||
});
|
||||
expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/EMFILE/);
|
||||
spy.mockRestore();
|
||||
|
||||
expect(() => sink.finalize()).toThrow(/IO error|EMFILE/);
|
||||
});
|
||||
|
||||
it('surfaces a final-flush IO failure from finalize (rows buffered, never mid-flushed)', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1000); // big chunk → no mid flush
|
||||
sink.addNode(bbNode('a.ts', 0, 1)); // buffered only (1 < 1000)
|
||||
|
||||
// The only writeSync happens during the final flush inside finalize → close.
|
||||
const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => {
|
||||
throw new Error('ENOSPC: disk full at close');
|
||||
});
|
||||
// close() never throws (it records poison); finalize reports it.
|
||||
expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/);
|
||||
spy.mockRestore();
|
||||
});
|
||||
|
||||
it('a poisoned writer stops accepting rows (no unbounded buffering on a dead fd)', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1);
|
||||
|
||||
const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => {
|
||||
throw new Error('ENOSPC');
|
||||
});
|
||||
expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/ENOSPC/); // poisons the writer
|
||||
// Subsequent rows are dropped silently at the writer (it is dead) — they do
|
||||
// not re-throw and do not accumulate; the run still fails at finalize.
|
||||
expect(() => sink.addNode(bbNode('a.ts', 1, 2))).not.toThrow();
|
||||
expect(() => sink.addNode(bbNode('a.ts', 2, 3))).not.toThrow();
|
||||
spy.mockRestore();
|
||||
|
||||
expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/);
|
||||
});
|
||||
});
|
||||
176
gitnexus/test/unit/rel-pair-routing.test.ts
Normal file
176
gitnexus/test/unit/rel-pair-routing.test.ts
Normal file
|
|
@ -0,0 +1,176 @@
|
|||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { EventEmitter } from 'events';
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import { RelPairRouter, getNodeLabel } from '../../src/core/lbug/rel-pair-routing.js';
|
||||
|
||||
/**
|
||||
* Unit tests for RelPairRouter (#2203 U2) — the production per-pair emit path.
|
||||
*
|
||||
* Mirrors test/unit/rel-csv-split.test.ts: drives the router with an injected
|
||||
* mock WriteStream factory so the error, backpressure, and teardown paths are
|
||||
* exercised without LadybugDB or real disk streams. These paths are otherwise
|
||||
* unreachable in the integration suite (which only hits the no-backpressure
|
||||
* happy path), so this is the coverage for the router's failure modes.
|
||||
*/
|
||||
|
||||
// Controllable backpressure + error injection (same shape as the split oracle's mock).
|
||||
class MockWriteStream extends EventEmitter {
|
||||
public chunks: string[] = [];
|
||||
public destroyed = false;
|
||||
public ended = false;
|
||||
public blocked = false;
|
||||
public maxDrainListenersSeen = 0;
|
||||
// State flags + events so `stream/promises.finished(ws)` (used by the
|
||||
// router's close()) resolves against this mock instead of hanging.
|
||||
public writable = true;
|
||||
public writableEnded = false;
|
||||
public writableFinished = false;
|
||||
|
||||
write(chunk: string): boolean {
|
||||
this.chunks.push(chunk);
|
||||
const count = this.listenerCount('drain');
|
||||
if (count > this.maxDrainListenersSeen) this.maxDrainListenersSeen = count;
|
||||
return !this.blocked;
|
||||
}
|
||||
|
||||
end(cb?: (err?: Error) => void): this {
|
||||
this.ended = true;
|
||||
this.writableEnded = true;
|
||||
this.writableFinished = true;
|
||||
this.writable = false;
|
||||
if (cb) cb();
|
||||
queueMicrotask(() => {
|
||||
this.emit('finish');
|
||||
this.emit('close');
|
||||
});
|
||||
return this;
|
||||
}
|
||||
|
||||
destroy(): this {
|
||||
this.destroyed = true;
|
||||
return this;
|
||||
}
|
||||
|
||||
unblock(): void {
|
||||
this.blocked = false;
|
||||
this.emit('drain');
|
||||
}
|
||||
|
||||
triggerError(err: Error): void {
|
||||
this.emit('error', err);
|
||||
}
|
||||
}
|
||||
|
||||
const HEADER = '"from","to","type","confidence","reason","step"';
|
||||
const VALID = new Set<string>(['File', 'Function', 'Community', 'Process']);
|
||||
|
||||
const row = (from: string, to: string, type = 'CALLS'): string =>
|
||||
`"${from}","${to}","${type}",1.0,"auto",0`;
|
||||
|
||||
function mockFactory(streams: MockWriteStream[], opts?: { blocked?: boolean }) {
|
||||
return (() => {
|
||||
const ws = new MockWriteStream();
|
||||
if (opts?.blocked) ws.blocked = true;
|
||||
streams.push(ws);
|
||||
return ws;
|
||||
}) as unknown as (filePath: string) => import('fs').WriteStream;
|
||||
}
|
||||
|
||||
let tmpDir: string;
|
||||
|
||||
beforeEach(() => {
|
||||
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'rel-pair-routing-test-'));
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 50 });
|
||||
});
|
||||
|
||||
describe('getNodeLabel', () => {
|
||||
it('maps comm_/proc_ prefixes and otherwise splits on the first colon', () => {
|
||||
expect(getNodeLabel('comm_42')).toBe('Community');
|
||||
expect(getNodeLabel('proc_7')).toBe('Process');
|
||||
expect(getNodeLabel('Function:src/a.ts:f:1')).toBe('Function');
|
||||
expect(getNodeLabel('File:src/a.ts')).toBe('File');
|
||||
});
|
||||
});
|
||||
|
||||
describe('RelPairRouter', () => {
|
||||
it('routes valid edges to per-pair files (header first) and skips invalid-label edges', async () => {
|
||||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams));
|
||||
|
||||
const route = async (from: string, to: string) => {
|
||||
const p = router.route(from, to, row(from, to));
|
||||
if (p) await p;
|
||||
};
|
||||
await route('File:a', 'Function:a:f:1');
|
||||
await route('File:a', 'Function:a:g:2'); // same pair
|
||||
await route('Function:a:f:1', 'Function:a:g:2'); // different pair
|
||||
await route('Bogus:x', 'File:a'); // invalid FROM label → skipped
|
||||
await route('File:a', 'Bogus:y'); // invalid TO label → skipped (other branch)
|
||||
await router.close();
|
||||
|
||||
expect(router.skipped).toBe(2);
|
||||
expect(router.total).toBe(3);
|
||||
expect([...router.byPair.keys()].sort()).toEqual(['File|Function', 'Function|Function']);
|
||||
expect(router.byPair.get('File|Function')!.rows).toBe(2);
|
||||
// Header is the first chunk written to each pair stream.
|
||||
expect(streams[0].chunks[0]).toBe(HEADER + '\n');
|
||||
expect(streams.every((s) => s.ended)).toBe(true);
|
||||
});
|
||||
|
||||
it('returns a drain promise under backpressure and completes once unblocked', async () => {
|
||||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(
|
||||
tmpDir,
|
||||
HEADER,
|
||||
VALID,
|
||||
mockFactory(streams, { blocked: true }),
|
||||
);
|
||||
|
||||
const pending = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'));
|
||||
expect(pending).toBeInstanceOf(Promise); // header write hit backpressure
|
||||
streams[0].unblock();
|
||||
await pending;
|
||||
|
||||
expect(streams[0].maxDrainListenersSeen).toBeLessThanOrEqual(1);
|
||||
expect(streams[0].chunks[0]).toBe(HEADER + '\n');
|
||||
expect(router.total).toBe(1);
|
||||
});
|
||||
|
||||
it('on a stream error: route() throws the real error, lastError exposes it, close() rejects + destroys', async () => {
|
||||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams));
|
||||
|
||||
const first = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'));
|
||||
if (first) await first;
|
||||
|
||||
const err = new Error('EMFILE: too many open files');
|
||||
streams[0].triggerError(err);
|
||||
|
||||
// The next route surfaces the REAL error, not a generic AbortError.
|
||||
expect(() => router.route('File:a', 'Function:a:g:2', row('File:a', 'Function:a:g:2'))).toThrow(
|
||||
'EMFILE',
|
||||
);
|
||||
expect(router.lastError).toBe(err);
|
||||
await expect(router.close()).rejects.toThrow('EMFILE');
|
||||
expect(streams[0].destroyed).toBe(true);
|
||||
});
|
||||
|
||||
it('destroy() tears down every open pair stream', async () => {
|
||||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams));
|
||||
|
||||
const a = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'));
|
||||
if (a) await a;
|
||||
const b = router.route('Community:1', 'Community:2', row('Community:1', 'Community:2'));
|
||||
if (b) await b;
|
||||
|
||||
router.destroy();
|
||||
expect(streams.length).toBe(2);
|
||||
expect(streams.every((s) => s.destroyed)).toBe(true);
|
||||
});
|
||||
});
|
||||
99
gitnexus/test/unit/stream-pdg-emit-config.test.ts
Normal file
99
gitnexus/test/unit/stream-pdg-emit-config.test.ts
Normal file
|
|
@ -0,0 +1,99 @@
|
|||
/**
|
||||
* Streaming PDG-emit config gating (issue #2202 U3).
|
||||
*
|
||||
* Verifies:
|
||||
* - `resolveStreamPdgEmit` engages only when pdg + full-rebuild (force) + an
|
||||
* enable signal (explicit option OR GITNEXUS_STREAM_PDG_EMIT) all hold;
|
||||
* - `resolvePdgEmitChunkSize` prefers the explicit option, falls back to
|
||||
* GITNEXUS_PDG_EMIT_CHUNK_SIZE, else undefined;
|
||||
* - the memory-only streaming knobs are NOT stamped into RepoMeta.pdg, so
|
||||
* changing them never trips `pdgModeMismatch` (would force needless full
|
||||
* writebacks otherwise).
|
||||
*/
|
||||
import { describe, it, expect, afterEach, vi } from 'vitest';
|
||||
import {
|
||||
resolveStreamPdgEmit,
|
||||
resolvePdgEmitChunkSize,
|
||||
pdgModeMismatch,
|
||||
resolvePdgConfig,
|
||||
} from '../../src/core/run-analyze.js';
|
||||
|
||||
afterEach(() => {
|
||||
vi.unstubAllEnvs();
|
||||
});
|
||||
|
||||
describe('resolveStreamPdgEmit — gating', () => {
|
||||
it('engages only with pdg + force + an enable signal', () => {
|
||||
// explicit option
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: true, streamPdgEmit: true })).toBe(true);
|
||||
// env toggle
|
||||
vi.stubEnv('GITNEXUS_STREAM_PDG_EMIT', '1');
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(true);
|
||||
});
|
||||
|
||||
it('does NOT engage without --force (incremental writeback reads BasicBlocks back)', () => {
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: false, streamPdgEmit: true })).toBe(false);
|
||||
expect(resolveStreamPdgEmit({ pdg: true, streamPdgEmit: true })).toBe(false);
|
||||
});
|
||||
|
||||
it('does NOT engage without --pdg (nothing to stream)', () => {
|
||||
expect(resolveStreamPdgEmit({ pdg: false, force: true, streamPdgEmit: true })).toBe(false);
|
||||
});
|
||||
|
||||
it('does NOT engage with no enable signal', () => {
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(false);
|
||||
vi.stubEnv('GITNEXUS_STREAM_PDG_EMIT', '0');
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('resolvePdgEmitChunkSize', () => {
|
||||
it('prefers the explicit option over the env var', () => {
|
||||
vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '128');
|
||||
expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 999 })).toBe(999);
|
||||
});
|
||||
|
||||
it('falls back to the env var, then to undefined', () => {
|
||||
vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '128');
|
||||
expect(resolvePdgEmitChunkSize({})).toBe(128);
|
||||
vi.unstubAllEnvs();
|
||||
expect(resolvePdgEmitChunkSize({})).toBeUndefined();
|
||||
});
|
||||
|
||||
it('rejects non-positive / non-integer env values (falls back to undefined)', () => {
|
||||
for (const bad of ['0', '-5', 'abc', '1.5', '']) {
|
||||
vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', bad);
|
||||
expect(resolvePdgEmitChunkSize({})).toBeUndefined();
|
||||
}
|
||||
});
|
||||
|
||||
it('rejects an explicit non-positive option (0/negative is not nullish — would defeat buffering)', () => {
|
||||
expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 0 })).toBeUndefined();
|
||||
expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: -10 })).toBeUndefined();
|
||||
// ...but an explicit 0 still falls back to a valid env value when present.
|
||||
vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '256');
|
||||
expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 0 })).toBe(256);
|
||||
});
|
||||
});
|
||||
|
||||
describe('streaming knobs are NOT emit-affecting (no pdgModeMismatch)', () => {
|
||||
it('changing streamPdgEmit / pdgEmitChunkSize does not trip pdgModeMismatch', () => {
|
||||
const base = { pdg: true as const };
|
||||
const recorded = resolvePdgConfig(base);
|
||||
// Same pdg config, but with the streaming knobs flipped — must NOT mismatch.
|
||||
expect(
|
||||
pdgModeMismatch(recorded, {
|
||||
...base,
|
||||
streamPdgEmit: true,
|
||||
pdgEmitChunkSize: 64,
|
||||
} as Parameters<typeof pdgModeMismatch>[1]),
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
it('streaming knobs are absent from the resolved RepoMeta.pdg stamp', () => {
|
||||
const stamp = resolvePdgConfig({ pdg: true });
|
||||
expect(stamp).toBeDefined();
|
||||
expect(stamp).not.toHaveProperty('streamPdgEmit');
|
||||
expect(stamp).not.toHaveProperty('pdgEmitChunkSize');
|
||||
});
|
||||
});
|
||||
16
package-lock.json
generated
16
package-lock.json
generated
|
|
@ -1868,10 +1868,20 @@
|
|||
"license": "MIT"
|
||||
},
|
||||
"node_modules/js-yaml": {
|
||||
"version": "4.1.1",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.1.1.tgz",
|
||||
"integrity": "sha512-qQKT4zQxXl8lLwBtHMWwaTcGfFOZviOJet3Oy/xmGk2gZH677CJM9EvtfdSkgWcATZhj/55JZ0rmy3myCT5lsA==",
|
||||
"version": "4.2.0",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.2.0.tgz",
|
||||
"integrity": "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/puzrin"
|
||||
},
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/nodeca"
|
||||
}
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"argparse": "^2.0.1"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue