diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index ee0d92332..4b0548e4f 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -276,6 +276,24 @@ jobs: run: node --expose-gc --import tsx bench/cfg/measure.mjs --check working-directory: gitnexus + - name: Emit-persistence throughput / byte-identity guards (#2203) + # Build-free: asserts streamAllCSVsToDisk output is byte-identical + # (order-independent CSV-line fingerprint — the #2203 U2/U3 emit + # optimisations must not change graph content) and that emit wall-time + # stays linear in node+edge count. The LadybugDB COPY half needs a real + # DB, so its timing lives in the runtime PROF_LBUG_LOAD breakdown. + run: node --import tsx bench/emit-persistence/measure.mjs --check + working-directory: gitnexus + + - name: Streaming PDG-emit byte-identity / bounded-RSS guards (#2202) + # Build-free: asserts the streaming PdgEmitSink emits a CSV row SET + # byte-identical to the whole-graph streamAllCSVsToDisk emit, AND that + # the in-memory graph retains zero BasicBlock nodes (the O(chunk) peak-RSS + # bound that unblocks full-kernel-scale repos). Fails on fingerprint drift + # or any resident BasicBlock. + run: node --import tsx bench/emit-persistence/measure-streaming.mjs --check + working-directory: gitnexus + - name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial) env: GITNEXUS_BENCH: '1' diff --git a/README.md b/README.md index ab01db810..55e5a03a7 100644 --- a/README.md +++ b/README.md @@ -324,6 +324,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max | `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. | | `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. | | `GITNEXUS_PROFILE_DEFERRED_SLOW_MS` | `3000` (verbose) / `5000` | Per-file threshold in ms above which `processCallsFromExtracted` emits a `slow file …` log line. Parsed via `Number()`: accepts integers (`5000`), scientific notation (`2.5e3`), decimals (`.5`), and hex (`0x10`). Non-finite or non-positive values fall back to the default. | Hunting a few outlier files dominating the deferred call-resolution stage; lower to surface more, raise to focus only on the worst. | +| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. | | `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size `. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. | | `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout ` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. | | `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold `. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. | diff --git a/gitnexus/bench/emit-persistence/README.md b/gitnexus/bench/emit-persistence/README.md new file mode 100644 index 000000000..0a690f40b --- /dev/null +++ b/gitnexus/bench/emit-persistence/README.md @@ -0,0 +1,58 @@ +# Emit-persistence bench (#2203) + +Build-free throughput + byte-identity guard for the **CSV-generation half** of +the graph-DB persistence pipeline (`streamAllCSVsToDisk`), which dominates +large-repo `analyze` wall time alongside parsing (issue #2203). + +```bash +# from gitnexus/ +node --import tsx bench/emit-persistence/measure.mjs # print one JSON line +node --import tsx bench/emit-persistence/measure.mjs --check # gate vs baselines.json +``` + +## What it measures + +A synthetic `KnowledgeGraph` (files + functions + classes + 4 edge types across +the `File→Function`, `File→Class`, `Function→Function` label pairs) at two +scales: + +- **`elapsed_ms_small` / `elapsed_ms_large`** — median wall-clock over `REPS` + runs of `streamAllCSVsToDisk`. +- **`scaling_ratio`** — `(t_large/t_small)/(LARGE/SMALL)`; ~1.0 is linear. The + `--check` gate fails if it exceeds `scaling_budget` (catches an O(n²) + re-regression in the emit/routing path). +- **`fingerprint`** — order-independent sha256 over every emitted CSV line (node + CSVs + per-FROM→TO-label-pair rel CSVs). This is the **byte-identity gate**: + the U2 (direct per-pair routing) and U3 (per-row microtask elimination) + optimisations must not change graph content, and any future change that does + fails `--check`. Byte-identity holds for all quote-free ids; for an id + containing a `"` the router intentionally diverges from — and is more correct + than — the legacy regex oracle (see `src/core/lbug/rel-pair-routing.ts`). + +## What it does NOT measure + +- **The LadybugDB `COPY` half.** Bulk loading needs a live writable DB + connection, so it can't run build-free. Its per-stage timing lives in the + runtime `PROF_LBUG_LOAD=1` breakdown (`[lbug-load prof] csv-emit=… copy-nodes=… + copy-rels=… fallback=… total=…`) and is exercised end-to-end by the + integration round-trip tests (`test/integration/basicblock-roundtrip.test.ts`, + `lbug-core-adapter.test.ts`). +- **Content extraction.** Bench nodes have no backing source files, so the + `content` column is empty — emit cost here reflects the CSV machinery + (routing, escaping, buffering, disk writes), not file reads. +- **At-scale absolute numbers.** The real postgres / kernel-`fs/` wall (issue + #2203's table) is a maintainer-run measurement; this synthetic bench is the + reproducible regression guard, not a substitute for those runs. + +## Deferred follow-up + +Parallelising the `COPY` loop (`PARALLEL=false` is load-bearing; LadybugDB is +single-writer) is **out of scope** for #2203 pending empirical validation of +concurrent-COPY support — the `PROF_LBUG_LOAD` breakdown is the prerequisite +that shows whether COPY is the dominant cost worth that risk. + +## Regenerating the baseline + +```bash +node --import tsx bench/emit-persistence/measure.mjs # copy fingerprint + ratio into baselines.json +``` diff --git a/gitnexus/bench/emit-persistence/baselines-streaming.json b/gitnexus/bench/emit-persistence/baselines-streaming.json new file mode 100644 index 000000000..62e1c0725 --- /dev/null +++ b/gitnexus/bench/emit-persistence/baselines-streaming.json @@ -0,0 +1,4 @@ +{ + "fingerprint": "386f432c74f4992455055d8891dbe6c873ea95afa60ef4be023a21aed7b4bcb1", + "_note": "Byte-identity + bounded-retention gate for streaming/chunked PDG emit (#2202). fingerprint = sha256 of the sorted, header-stripped BasicBlock + PDG-edge data rows of the canonical synthetic set. --check also asserts the streamed PdgEmitSink output is byte-identical to the whole-graph streamAllCSVsToDisk emit (byte_identical_nodes/edges) and that the in-memory graph retains 0 BasicBlocks (resident_basic_blocks === 0, the O(chunk) RSS bound). Regenerate via `node --import tsx bench/emit-persistence/measure-streaming.mjs`." +} diff --git a/gitnexus/bench/emit-persistence/baselines.json b/gitnexus/bench/emit-persistence/baselines.json new file mode 100644 index 000000000..93806b252 --- /dev/null +++ b/gitnexus/bench/emit-persistence/baselines.json @@ -0,0 +1,6 @@ +{ + "fingerprint": "1b9dd0b783899b47067c36511d241860f291ac736e57682b0ece14148e3958ff", + "scaling_budget": 1.8, + "max_ms_large": 1000, + "_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted — binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." +} diff --git a/gitnexus/bench/emit-persistence/measure-streaming.mjs b/gitnexus/bench/emit-persistence/measure-streaming.mjs new file mode 100644 index 000000000..683b86888 --- /dev/null +++ b/gitnexus/bench/emit-persistence/measure-streaming.mjs @@ -0,0 +1,199 @@ +/** + * Build-free byte-identity + bounded-retention bench for streaming/chunked PDG + * graph emit (issue #2202). + * + * Proves the two acceptance criteria at scale, without a DB connection: + * 1. BYTE-IDENTITY (R2): emitting a BasicBlock + intra-file PDG-edge set via + * the streaming `PdgEmitSink` produces the IDENTICAL CSV data-row set as + * the whole-graph `streamAllCSVsToDisk` path. Compared per file + * (basicblock.csv, rel_BasicBlock_BasicBlock.csv) over header-stripped, + * sorted lines so it is a pure function of the emitted row SET. + * 2. BOUNDED RETENTION (R1): with streaming on, the in-memory graph holds + * ZERO BasicBlock nodes regardless of how many are emitted — the PDG layer + * never accumulates in process memory (peak RSS O(chunk), not O(graph)). + * + * Build-free: imports the `.ts` hotpaths through tsx + * (`node --import tsx bench/emit-persistence/measure-streaming.mjs`). + * + * Without args: prints one JSON object. With `--check`: asserts byte-identity, + * retention, and fingerprint == the committed baseline; exits non-zero on any + * failure. + */ +import fs from 'node:fs'; +import fsp from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { fileURLToPath } from 'node:url'; + +import { createKnowledgeGraph } from '../../src/core/graph/graph.ts'; +import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.ts'; +import { PdgEmitSink } from '../../src/core/lbug/pdg-emit-sink.ts'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines-streaming.json'); + +// PDG edge types streamed per file (all intra-block BasicBlock→BasicBlock). +const PDG_TYPES = ['CFG', 'REACHING_DEF', 'CDG', 'POST_DOMINATE', 'TAINTED', 'SANITIZES']; + +const FUNCS = 1200; // functions +const BLOCKS = 6; // basic blocks per function ⇒ FUNCS*BLOCKS BasicBlocks total +const CHUNK_ROWS = 64; // tiny streamed buffer to exercise frequent flushing + +/** + * Build the canonical PDG node/edge SET: `FUNCS` functions each with `BLOCKS` + * BasicBlocks and a chain of intra-function PDG edges. Returns the structural + * nodes (File/Function) separately from the BasicBlock + PDG-edge layer so the + * streamed path can route them to different sinks. + */ +function buildSet() { + const structuralNodes = []; + const structuralRels = []; + const bbNodes = []; + const pdgEdges = []; + for (let f = 0; f < FUNCS; f++) { + const fp = `src/m${f % 50}.ts`; + const fnId = `Function:${fp}:fn${f}:1`; + structuralNodes.push({ + id: fnId, + label: 'Function', + properties: { name: `fn${f}`, filePath: fp, startLine: 1, endLine: 99 }, + }); + for (let b = 0; b < BLOCKS; b++) { + bbNodes.push({ + id: `BasicBlock:${fp}:1:0:${f}_${b}`, + label: 'BasicBlock', + properties: { + name: '', + filePath: fp, + startLine: b * 3, + endLine: b * 3 + 2, + text: `f${f}b${b}`, + }, + }); + } + for (let b = 0; b < BLOCKS - 1; b++) { + const from = `BasicBlock:${fp}:1:0:${f}_${b}`; + const to = `BasicBlock:${fp}:1:0:${f}_${b + 1}`; + for (const type of PDG_TYPES) { + pdgEdges.push({ + id: `${type}:${f}:${b}`, + sourceId: from, + targetId: to, + type, + confidence: 1, + reason: type === 'REACHING_DEF' ? `v${b}` : type === 'CDG' ? 'T' : '', + }); + } + } + } + // A few File nodes so the structural emit produces a realistic multi-table mix. + for (let m = 0; m < 50; m++) { + structuralNodes.push({ + id: `File:src/m${m}.ts`, + label: 'File', + properties: { name: `m${m}.ts`, filePath: `src/m${m}.ts` }, + }); + } + return { structuralNodes, structuralRels, bbNodes, pdgEdges }; +} + +/** Header-stripped, sorted, non-empty data rows of one CSV file (or [] if absent). */ +async function dataRows(csvPath) { + let text; + try { + text = await fsp.readFile(csvPath, 'utf8'); + } catch { + return []; + } + const lines = text.split('\n').filter((l) => l.length > 0); + return lines.slice(1).sort(); // drop the header line +} + +const sha = (rows) => crypto.createHash('sha256').update(rows.join('\n')).digest('hex'); + +async function measure() { + // mkdtemp (unpredictable, unique) rather than a predictable pid-based tmp path. + const tmpRoot = await fsp.mkdtemp(path.join(os.tmpdir(), 'gitnexus-stream-bench-')); + try { + const { structuralNodes, structuralRels, bbNodes, pdgEdges } = buildSet(); + + // ── whole-graph path ───────────────────────────────────────────────── + const wholeGraph = createKnowledgeGraph(); + for (const n of structuralNodes) wholeGraph.addNode(n); + for (const n of bbNodes) wholeGraph.addNode(n); + for (const r of structuralRels) wholeGraph.addRelationship(r); + for (const e of pdgEdges) wholeGraph.addRelationship(e); + const wholeDir = path.join(tmpRoot, 'whole'); + await streamAllCSVsToDisk(wholeGraph, path.join(tmpRoot, 'no-repo'), wholeDir); + + // ── streamed path ──────────────────────────────────────────────────── + const realGraph = createKnowledgeGraph(); + const sink = new PdgEmitSink(realGraph, path.join(tmpRoot, 'pdg-csv'), CHUNK_ROWS); + for (const n of structuralNodes) realGraph.addNode(n); // structural → real graph + for (const r of structuralRels) realGraph.addRelationship(r); + for (const n of bbNodes) sink.addNode(n); // BasicBlock layer → sink (CSV) + for (const e of pdgEdges) sink.addRelationship(e); + sink.finalize(); + const streamedCsvDir = path.join(tmpRoot, 'streamed'); + await streamAllCSVsToDisk(realGraph, path.join(tmpRoot, 'no-repo'), streamedCsvDir); + + // ── retention (R1): the real graph holds ZERO BasicBlocks ──────────── + let residentBasicBlocks = 0; + for (const n of realGraph.iterNodes()) if (n.label === 'BasicBlock') residentBasicBlocks++; + + // ── byte-identity (R2): per-file data-row set equality ─────────────── + const wholeBb = await dataRows(path.join(wholeDir, 'basicblock.csv')); + const streamedBb = await dataRows(path.join(tmpRoot, 'pdg-csv', 'basicblock.csv')); + const wholeRel = await dataRows(path.join(wholeDir, 'rel_BasicBlock_BasicBlock.csv')); + const streamedRel = await dataRows( + path.join(tmpRoot, 'pdg-csv', 'rel_BasicBlock_BasicBlock.csv'), + ); + + const bbIdentical = sha(wholeBb) === sha(streamedBb); + const relIdentical = sha(wholeRel) === sha(streamedRel); + // Fingerprint over the canonical PDG data-row set (drift gate). + const fingerprint = sha([...wholeBb, ...wholeRel].sort()); + + return { + scenario: 'streamingPdgEmit', + basic_blocks: bbNodes.length, + pdg_edges: pdgEdges.length, + chunk_rows: CHUNK_ROWS, + resident_basic_blocks: residentBasicBlocks, + byte_identical_nodes: bbIdentical, + byte_identical_edges: relIdentical, + fingerprint, + }; + } finally { + await fsp.rm(tmpRoot, { recursive: true, force: true }).catch(() => {}); + } +} + +const CHECK = process.argv.includes('--check'); +const result = await measure(); + +if (!CHECK) { + process.stdout.write(JSON.stringify(result) + '\n'); +} else { + const base = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8')); + const failures = []; + if (!result.byte_identical_nodes) + failures.push('streamed BasicBlock rows differ from whole-graph emit'); + if (!result.byte_identical_edges) + failures.push('streamed PDG-edge rows differ from whole-graph emit'); + if (result.resident_basic_blocks !== 0) { + failures.push( + `RSS bound violated: ${result.resident_basic_blocks} BasicBlock node(s) retained in the in-memory graph (expected 0)`, + ); + } + if (result.fingerprint !== base.fingerprint) { + failures.push(`fingerprint drift (got ${result.fingerprint}, expected ${base.fingerprint})`); + } + process.stdout.write(JSON.stringify(result) + '\n'); + if (failures.length > 0) { + for (const f of failures) process.stderr.write(`[stream-pdg-emit --check] FAIL: ${f}\n`); + process.exit(1); + } + process.stderr.write('[stream-pdg-emit --check] PASS\n'); +} diff --git a/gitnexus/bench/emit-persistence/measure.mjs b/gitnexus/bench/emit-persistence/measure.mjs new file mode 100644 index 000000000..f7623bc34 --- /dev/null +++ b/gitnexus/bench/emit-persistence/measure.mjs @@ -0,0 +1,216 @@ +/** + * Build-free emit-path throughput + byte-identity bench for the graph-DB + * persistence pipeline (issue #2203). + * + * Measures `streamAllCSVsToDisk` — the CSV-generation half of the persistence + * path that U2 (direct per-pair relationship routing) and U3 (per-row + * microtask elimination) optimised. The LadybugDB `COPY` half needs a real DB + * connection, so its timing lives in the runtime `PROF_LBUG_LOAD` breakdown + + * the integration round-trip tests, NOT here (see README.md). + * + * For a synthetic KnowledgeGraph at two scales it reports: + * - elapsed_ms_small / elapsed_ms_large (median over REPS) + a scaling ratio + * `(t_large/t_small)/(LARGE/SMALL)`: ~1.0 linear, ~3.x quadratic; + * - an order-independent sha256 fingerprint over every emitted CSV line + * (node CSVs + per-FROM→TO-label-pair rel CSVs), as the byte-identity gate + * guarding the issue's "byte-identical graph content" requirement. + * + * Build-free: imports the `.ts` hotpaths through tsx + * (`node --import tsx bench/emit-persistence/measure.mjs`). Static `.ts` + * imports work; a top-level `await import()` breaks tsx's lexer. + * + * Without args: prints one JSON object per scenario. + * With `--check`: asserts the fingerprint == the committed baseline AND the + * scaling ratio < the recorded budget; exits non-zero on drift/regression. + */ +import fs from 'node:fs'; +import fsp from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { fileURLToPath } from 'node:url'; + +import { createKnowledgeGraph } from '../../src/core/graph/graph.ts'; +import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.ts'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +// ---- synthetic graph generation (deterministic — no randomness) ---- + +/** + * Build a graph of `entityCount` files, each with 2 functions + 1 class and 4 + * relationships. ids carry valid table-label prefixes (File:/Function:/Class:) + * so edges route to real pairs (File→Function, File→Class, Function→Function) + * — exercising the U2 router across multiple label pairs. Node `content` is + * never populated (no backing files), so emit cost reflects the CSV machinery + * (routing, escaping, buffering, disk writes), not content extraction. + */ +function generateGraph(entityCount) { + const graph = createKnowledgeGraph(); + for (let i = 0; i < entityCount; i++) { + const fp = `src/e${i}.ts`; + const fileId = `File:${fp}`; + const fnA = `Function:${fp}:fnA:1`; + const fnB = `Function:${fp}:fnB:10`; + const cls = `Class:${fp}:C:20`; + graph.addNode({ id: fileId, label: 'File', properties: { name: `e${i}.ts`, filePath: fp } }); + graph.addNode({ + id: fnA, + label: 'Function', + properties: { name: 'fnA', filePath: fp, startLine: 1, endLine: 5, isExported: true }, + }); + graph.addNode({ + id: fnB, + label: 'Function', + properties: { name: 'fnB', filePath: fp, startLine: 10, endLine: 15, isExported: false }, + }); + graph.addNode({ + id: cls, + label: 'Class', + properties: { name: 'C', filePath: fp, startLine: 20, endLine: 30, isExported: true }, + }); + graph.addRelationship({ + id: `${fileId}->${fnA}`, + sourceId: fileId, + targetId: fnA, + type: 'CONTAINS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: `${fileId}->${fnB}`, + sourceId: fileId, + targetId: fnB, + type: 'CONTAINS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: `${fileId}->${cls}`, + sourceId: fileId, + targetId: cls, + type: 'CONTAINS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: `${fnA}->${fnB}`, + sourceId: fnA, + targetId: fnB, + type: 'CALLS', + confidence: 1, + reason: '', + }); + } + return graph; +} + +// ---- byte-identity fingerprint (order-independent) ---- + +/** sha256 over every non-empty line of every emitted CSV file, sorted so the + * digest is a pure function of the emitted line SET (insertion-order agnostic). */ +async function fingerprintEmit(graph, dir) { + await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); + // Per-file digest bound to the filename: a row routed to the WRONG pair file + // (or a header written to the wrong file) changes the fingerprint — a global + // line-flatten could not catch that. File bytes are hashed as-written (so it + // also catches within-file row reordering); the entry list is sorted so + // readdir order doesn't matter. + const entries = []; + for (const name of fs.readdirSync(dir)) { + if (!name.endsWith('.csv')) continue; + const bytes = await fsp.readFile(path.join(dir, name)); + entries.push(`${name}\n${crypto.createHash('sha256').update(bytes).digest('hex')}`); + } + return crypto.createHash('sha256').update(entries.sort().join('\n')).digest('hex'); +} + +// ---- timing ---- + +function median(xs) { + const s = [...xs].sort((a, b) => a - b); + const m = Math.floor(s.length / 2); + return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2; +} + +async function timeEmit(graph, dir, reps) { + await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); // warmup (not counted) + const samples = []; + for (let i = 0; i < reps; i++) { + const start = process.hrtime.bigint(); + await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); + samples.push(Number(process.hrtime.bigint() - start) / 1e6); + } + return median(samples); +} + +const SMALL = 600; +const LARGE = 2400; +const REPS = 5; + +async function measure() { + const tmpRoot = path.join(os.tmpdir(), `gitnexus-emit-bench-${process.pid}`); + await fsp.mkdir(tmpRoot, { recursive: true }); + try { + const smallGraph = generateGraph(SMALL); + const largeGraph = generateGraph(LARGE); + + const fingerprint = await fingerprintEmit(largeGraph, path.join(tmpRoot, 'fp')); + const small = await timeEmit(smallGraph, path.join(tmpRoot, 'small'), REPS); + const large = await timeEmit(largeGraph, path.join(tmpRoot, 'large'), REPS); + const scalingRatio = small > 0 ? large / small / (LARGE / SMALL) : 0; + + return { + scenario: 'streamAllCSVsToDisk', + entities_small: SMALL, + entities_large: LARGE, + nodes_large: LARGE * 4, + rels_large: LARGE * 4, + elapsed_ms_small: Number(small.toFixed(2)), + elapsed_ms_large: Number(large.toFixed(2)), + scaling_ratio: Number(scalingRatio.toFixed(3)), + fingerprint, + }; + } finally { + await fsp.rm(tmpRoot, { recursive: true, force: true }).catch(() => {}); + } +} + +// ---- run ---- + +const CHECK = process.argv.includes('--check'); +const result = await measure(); + +if (!CHECK) { + process.stdout.write(JSON.stringify(result) + '\n'); +} else { + const base = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8')); + const failures = []; + if (result.fingerprint !== base.fingerprint) { + failures.push( + `byte-identity fingerprint drift (got ${result.fingerprint}, expected ${base.fingerprint})`, + ); + } + if (result.scaling_ratio >= base.scaling_budget) { + failures.push( + `scaling ratio ${result.scaling_ratio} >= budget ${base.scaling_budget} ` + + `(${SMALL}->${LARGE} entities, ms ${result.elapsed_ms_small}->${result.elapsed_ms_large})`, + ); + } + // Absolute backstop: the scaling ratio alone passes a uniform Nx slowdown (it + // only compares large/small). A generous, host-noise-tolerant ceiling catches + // a gross absolute regression. Opt-in (only enforced when max_ms_large is set). + if (base.max_ms_large !== undefined && result.elapsed_ms_large >= base.max_ms_large) { + failures.push( + `absolute wall-time regression: elapsed_ms_large ${result.elapsed_ms_large}ms >= budget ` + + `${base.max_ms_large}ms (coarse backstop, not a tight SLA)`, + ); + } + process.stdout.write(JSON.stringify(result) + '\n'); + if (failures.length > 0) { + for (const f of failures) process.stderr.write(`[emit-persistence --check] FAIL: ${f}\n`); + process.exit(1); + } + process.stderr.write('[emit-persistence --check] PASS\n'); +} diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 991729ee6..3a93b9339 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -1340,19 +1340,18 @@ "license": "BSD-3-Clause" }, "node_modules/@protobufjs/eventemitter": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.0.tgz", - "integrity": "sha512-j9ednRT81vYJ9OfVuXG6ERSTdEL1xVsNgqpkxMsbIabzSo3goCjDIveeGv5d03om39ML71RdmrGNjG5SReBP/Q==", + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.1.tgz", + "integrity": "sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg==", "license": "BSD-3-Clause" }, "node_modules/@protobufjs/fetch": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.0.tgz", - "integrity": "sha512-lljVXpqXebpsijW71PZaCYeIcE5on1w5DlQy5WH6GLbFryLUrBD4932W/E2BSpfRJWseIL4v/KPgBFxDOIdKpQ==", + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.1.tgz", + "integrity": "sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw==", "license": "BSD-3-Clause", "dependencies": { - "@protobufjs/aspromise": "^1.1.1", - "@protobufjs/inquire": "^1.1.0" + "@protobufjs/aspromise": "^1.1.1" } }, "node_modules/@protobufjs/float": { @@ -1361,12 +1360,6 @@ "integrity": "sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ==", "license": "BSD-3-Clause" }, - "node_modules/@protobufjs/inquire": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@protobufjs/inquire/-/inquire-1.1.1.tgz", - "integrity": "sha512-mnzgDV26ueAvk7rsbt9L7bE0SuAoqyuys/sMMrmVcN5x9VsxpcG3rqAUSgDyLp0UZlmNfIbQ4fHfCtreVBk8Ew==", - "license": "BSD-3-Clause" - }, "node_modules/@protobufjs/path": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/@protobufjs/path/-/path-1.1.2.tgz", @@ -1822,9 +1815,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "25.9.2", - "resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.2.tgz", - "integrity": "sha512-G05zqtJhcDLb8uslf5EjCxXg9G1KQxiV8OS0R26IC//Eoyitzqe8z37I7cqvnZlrlSfgocQRfSn/AHBZJJFyGw==", + "version": "25.9.3", + "resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.3.tgz", + "integrity": "sha512-603BddQMv3pUcr4U2dhujk83N2tTDVr/34wII2B6bJy6g+8WD6yUb11jszNs0gdi4PesVWl7ABt8nYMVpnLUcg==", "license": "MIT", "dependencies": { "undici-types": ">=7.24.0 <7.24.7" @@ -4351,24 +4344,23 @@ "license": "MIT" }, "node_modules/protobufjs": { - "version": "7.5.8", - "resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.5.8.tgz", - "integrity": "sha512-dvpCIeLPbXZS/Ete7yLaO7RenOdken2NHKykBXbsaGxZT0UTltcarBciw+A78SRQs9iMAAVpsYA+l8b1hTePIA==", + "version": "7.6.4", + "resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.6.4.tgz", + "integrity": "sha512-RJJPTTpvFfHcWLkIa2JFWK4XvtSzS0yEWDmunqHXli1h3JlkbcQZXDZdcWxv+JK3Xsl5/UFDPZ0iGm7DAengYw==", "hasInstallScript": true, "license": "BSD-3-Clause", "dependencies": { "@protobufjs/aspromise": "^1.1.2", "@protobufjs/base64": "^1.1.2", "@protobufjs/codegen": "^2.0.5", - "@protobufjs/eventemitter": "^1.1.0", - "@protobufjs/fetch": "^1.1.0", + "@protobufjs/eventemitter": "^1.1.1", + "@protobufjs/fetch": "^1.1.1", "@protobufjs/float": "^1.0.2", - "@protobufjs/inquire": "^1.1.1", "@protobufjs/path": "^1.1.2", "@protobufjs/pool": "^1.1.0", "@protobufjs/utf8": "^1.1.1", "@types/node": ">=13.7.0", - "long": "^5.0.0" + "long": "^5.3.2" }, "engines": { "node": ">=12.0.0" @@ -4917,9 +4909,9 @@ } }, "node_modules/tar": { - "version": "7.5.13", - "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.13.tgz", - "integrity": "sha512-tOG/7GyXpFevhXVh8jOPJrmtRpOTsYqUIkVdVooZYJS/z8WhfQUX8RJILmeuJNinGAMSu1veBr4asSHFt5/hng==", + "version": "7.5.16", + "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz", + "integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==", "license": "BlueOak-1.0.0", "dependencies": { "@isaacs/fs-minipass": "^4.0.0", diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index 334c022ac..ff01ccdb6 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -121,6 +121,23 @@ export interface PipelineOptions { /** Per-run `TAINT_PATH` edge cap (#2084 review P1-3). `undefined` ⇒ * `DEFAULT_PDG_MAX_INTERPROC_EDGES` (1000); `0` ⇒ no cap. */ pdgMaxInterprocEdges?: number; + /** + * Streaming/chunked PDG graph emit (#2202). When true, the BasicBlock + + * intra-file PDG-edge layer (CFG / REACHING_DEF / CDG / POST_DOMINATE / + * TAINTED / SANITIZES) is streamed to CSV-on-disk during the scope-resolution + * emit loop instead of being materialized in the in-memory graph, bounding + * peak RSS to O(chunk) rather than O(graph) at full-kernel scale. Already + * gated by the caller to full-rebuild runs only (the incremental writeback + * reads BasicBlocks back from the in-memory graph). Memory-only — produces a + * byte-identical persisted graph and is NOT part of `RepoMeta.pdg`, so + * toggling it never trips `pdgModeMismatch`. Default/false ⇒ today's + * whole-graph emit. + */ + streamPdgEmit?: boolean; + /** Streamed PDG-emit write buffer (rows) when `streamPdgEmit` is on (#2202). + * `undefined` ⇒ `DEFAULT_PDG_EMIT_CHUNK_ROWS`. Memory-only; does not affect + * emitted bytes. */ + pdgEmitChunkSize?: number; /** * Request parsing with the worker pool disabled. The sequential parser was * removed — the worker pool is the sole parse path — so setting this now @@ -287,10 +304,10 @@ export const runPipelineFromRepo = async ( let communityResult: CommunitiesOutput['communityResult'] | undefined; let processResult: ProcessesOutput['processResult'] | undefined; - const resolutionOutcomes = getPhaseOutput( - results, - 'scopeResolution', - ).resolutionOutcomes; + const scopeResolutionOutput = getPhaseOutput(results, 'scopeResolution'); + const resolutionOutcomes = scopeResolutionOutput.resolutionOutcomes; + // Streamed PDG-emit manifest (#2202): present only when streaming was on. + const pdgEmitManifest = scopeResolutionOutput.pdgEmitManifest; if (!options?.skipGraphPhases) { communityResult = getPhaseOutput(results, 'communities').communityResult; @@ -319,5 +336,6 @@ export const runPipelineFromRepo = async ( processResult, resolutionOutcomes, usedWorkerPool, + pdgEmitManifest, }; }; diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index 6bf86dc84..c7671d8f4 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -44,6 +44,8 @@ import { import type { ResolutionOutcome } from '../resolution-outcome.js'; import type { FunctionSummary } from '../../taint/summary-model.js'; import { buildFunctionNodeIndex } from '../../taint/summary-harvest-driver.js'; +import { PdgEmitSink, type PdgEmitManifest } from '../../../lbug/pdg-emit-sink.js'; +import { resolveNativeSafeStorageDir } from '../../../lbug/lbug-config.js'; import { logger } from '../../../logger.js'; export interface ScopeResolutionOutput { @@ -72,6 +74,14 @@ export interface ScopeResolutionOutput { * The `taintSummaries` phase composes these over the `CALLS` graph. */ readonly functionSummaries: readonly FunctionSummary[]; + /** + * Streamed PDG-emit COPY manifest (#2202). Present only when streaming was on + * (full rebuild + `--pdg` + enabled): the BasicBlock node CSV + per-pair PDG + * edge CSVs that were flushed to disk during the emit loop, for the persistence + * step to COPY alongside the structural CSVs. Absent ⇒ the PDG layer (if any) + * is in the in-memory graph and persists via the normal whole-graph emit. + */ + readonly pdgEmitManifest?: PdgEmitManifest; } const NOOP_OUTPUT: ScopeResolutionOutput = Object.freeze({ @@ -242,236 +252,291 @@ export const scopeResolutionPhase: PipelinePhase = { ? buildFunctionNodeIndex(ctx.graph) : undefined; - for (const [lang, provider] of SCOPE_RESOLVERS) { - // Standalone providers (COBOL, JCL) don't emit graph edges yet - // through the scope-resolution path. This is the canonical guard: - // runScopeResolution is never called for standalone providers, which - // keeps cobolPhase as the sole IMPORTS edge producer. Keep this guard - // in sync with any additional standalone providers added to - // SCOPE_RESOLVERS. - if (provider.languageProvider.parseStrategy === 'standalone') continue; - - const primaryLangFiles = filesByLang.get(lang) ?? []; - if (primaryLangFiles.length === 0) continue; - const primaryFilePaths = primaryLangFiles.map((f) => f.path); - - // Load per-language import-resolution config (tsconfig paths, - // composer.json autoload, go.mod, ...). One I/O round trip per - // workspace pass — cached implicitly by the result handed to - // every `resolveImportTarget` call below. - const resolutionConfig = - provider.loadResolutionConfig !== undefined - ? await provider.loadResolutionConfig(ctx.repoPath) - : undefined; - - // Some languages (e.g. Vue) expand their file universe beyond the - // primary-language files via the `collectScopeContextPaths` hook. - // The hook receives raw source contents of the primary files so it - // can trace import closures without a second tree-sitter parse. - // - // To avoid reading primary files twice (once for the hook, once for - // the resolution pass), we read them upfront and merge with the - // extra context paths the hook may add. - // Stream this language's pre-built ParsedFiles in from the disk store - // FIRST (huge-repo path). Doing it before reading source lets us skip - // loading content for files the store already covers — for a provider - // with no content-consuming hook that source is pure dead weight once - // extraction is served from the store (~1.5 GB on the kernel's C pass). - // Merged into `preExtractedByPath`; the per-language release block below - // evicts these again before the next language, so only one language's - // ParsedFiles are resident at a time. - const loadStoreFor = async (paths: ReadonlySet): Promise => { - if (!parsedFileStorePath) return; - const fromDisk = await loadParsedFilesForPaths(parsedFileStorePath, paths); - for (const [fp, pf] of fromDisk) preExtractedByPath.set(fp, pf); - }; - - // A provider that feeds source text into a post-extract hook - // (populateWorkspaceOwners / populateNamespaceSiblings / - // populateRangeBindings / emitPostResolutionEdges) needs content for ALL - // its files; one without those hooks only needs content for files the - // store does NOT cover (fresh-extract fallback). Keep this in sync with - // the getFileContents() call-sites in run.ts. - const providerNeedsAllContent = - provider.populateWorkspaceOwners !== undefined || - provider.populateNamespaceSiblings !== undefined || - provider.populateRangeBindings !== undefined || - provider.emitPostResolutionEdges !== undefined; - - let scopeFilePaths: Set; - let contents: Map; - if (provider.collectScopeContextPaths !== undefined) { - // Context-expanding providers (e.g. Vue) need every primary file's - // source up front for the closure hook, so load it all. - const entryFileContents = await readFileContents(ctx.repoPath, primaryFilePaths); - scopeFilePaths = provider.collectScopeContextPaths({ - primaryFilePaths, - preExtractedByPath, - entryFileContents, - allScannedPaths, - resolutionConfig, - }); - // Read only the extra context files (TS/JS etc.) not already loaded. - const extraPaths = [...scopeFilePaths].filter((p) => !entryFileContents.has(p)); - const extraContents = await readFileContents(ctx.repoPath, extraPaths); - contents = new Map([...entryFileContents, ...extraContents]); - await loadStoreFor(scopeFilePaths); + // Streaming/chunked PDG emit (#2202): when enabled (the caller has already + // gated this to full-rebuild + `--pdg`), route the BasicBlock + intra-file + // PDG-edge layer to CSV-on-disk through one sink shared across every + // language pass, so it never accumulates in `ctx.graph` (peak RSS O(chunk)). + // Needs the storage dir (the parse-cache store path, the same `.gitnexus` + // dir loadGraphToLbug COPYs from); if that is somehow absent we skip + // streaming and fall back to the in-memory whole-graph emit. + let pdgEmitSink: PdgEmitSink | undefined; + if (ctx.options?.streamPdgEmit === true && totalScopeFiles > 0) { + if (parsedFileStorePath) { + pdgEmitSink = new PdgEmitSink( + ctx.graph, + // Same ASCII-safe relocation the structural CSVs get (#2202 review #2): + // on Windows non-ASCII storage paths the COPY can't open files under + // the native path, so the dir is relocated to a hashed os.tmpdir(). + resolveNativeSafeStorageDir(parsedFileStorePath, 'pdg-csv'), + ctx.options?.pdgEmitChunkSize, + ); } else { - scopeFilePaths = new Set(primaryFilePaths); - await loadStoreFor(scopeFilePaths); - const pathsToRead = providerNeedsAllContent - ? primaryFilePaths - : primaryFilePaths.filter((p) => !preExtractedByPath.has(p)); - contents = await readFileContents(ctx.repoPath, pathsToRead); - } - const filePaths = [...scopeFilePaths]; - const files: { path: string; content: string }[] = []; - for (const fp of filePaths) { - const content = contents.get(fp); - if (content !== undefined) { - files.push({ path: fp, content }); - } else if (preExtractedByPath.has(fp)) { - // Store covers extraction for this file and we deliberately skipped - // reading its source; the empty string is never consumed (the - // extract loop uses the pre-extracted ParsedFile and this provider - // has no content hook). - files.push({ path: fp, content: '' }); - } - // else: uncovered AND unreadable → skip (unchanged from prior behavior). - } - - const langFileCount = files.length; - logHeapProbe( - 'scope-lang-start', - `lang=${lang} files=${langFileCount} contentsLoaded=${contents.size}`, - ); - const langLabel = lang.charAt(0).toUpperCase() + lang.slice(1); - currentLangIdx++; - const langTag = - totalScopeLangs > 1 ? `${langLabel} [${currentLangIdx}/${totalScopeLangs}]` : langLabel; - - if (totalScopeFiles > 0) { - const pct = - SCOPE_PCT_START + Math.round((processedScopeFiles / totalScopeFiles) * SCOPE_PCT_RANGE); - ctx.onProgress({ - phase: 'scopeResolution', - percent: pct, - message: 'Resolving types', - detail: `${langTag}, ${langFileCount.toLocaleString()} files`, - }); - } - - const stats = runScopeResolution( - { - graph: ctx.graph, - model, - files, - resolutionConfig, - prebuiltNodeLookup: sharedNodeLookup, - prebuiltFunctionNodeIndex: sharedFnNodeIndex, - preExtractedParsedFiles: preExtractedByPath, - scopeIndexStorePath: parsedFileStorePath, - // CFG/PDG emission (#2081 M1) — opt-in; off ⇒ byte-identical graph. - pdg: ctx.options?.pdg === true, - pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction, - pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction, - pdgMaxCdgEdgesPerFunction: ctx.options?.pdgMaxCdgEdgesPerFunction, - pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction, - pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops, - recordResolutionOutcome: (outcome) => { - resolutionOutcomes.push(outcome); - }, - onWarn: (msg) => { - if (isSemanticModelValidatorEnabled()) { - logger.warn(`[scope-resolution:${lang}] ${msg}`); - } - }, - onProgress: - totalScopeFiles > 0 - ? (subPhase: ScopeResolutionSubPhase, current, total) => { - let langRatio: number; - switch (subPhase) { - case 'extracting': - langRatio = total > 0 ? (current / total) * 0.5 : 0; - break; - case 'analyzing types': - langRatio = 0.5; - break; - case 'resolving references': - langRatio = 0.7; - break; - case 'linking symbols': - langRatio = 0.85; - break; - default: { - const _exhaustive: never = subPhase; - langRatio = 0.85; - } - } - const overallRatio = Math.min( - 1, - (processedScopeFiles + langRatio * langFileCount) / totalScopeFiles, - ); - const pct = SCOPE_PCT_START + Math.round(overallRatio * SCOPE_PCT_RANGE); - ctx.onProgress({ - phase: 'scopeResolution', - percent: pct, - message: 'Resolving types', - detail: - subPhase === 'extracting' - ? `${langTag} — extracting ${current.toLocaleString()}/${total.toLocaleString()} files` - : `${langTag} — ${subPhase}`, - }); - } - : undefined, - }, - provider, - ); - - // Release file contents and pre-extracted entries after each language - // to reduce memory pressure. For large codebases (16K+ PHP files), - // holding all source code simultaneously with scope trees causes OOM. - // See: https://github.com/abhigyanpatwari/GitNexus/issues/1741 - // - // Use `filePaths` (not `primaryFilePaths`) so that any context files - // added by `collectScopeContextPaths` (e.g. TS/JS files pulled in for - // Vue cross-file resolution) are also evicted and not held until GC. - files.length = 0; - contents.clear(); - for (const fp of filePaths) { - preExtractedByPath.delete(fp); - } - // This language's ParsedFiles are now unreachable (runScopeResolution has - // returned and the Map entries are deleted). Force a GC HERE so a heavy - // language's ~17-20GB set (e.g. C/C++ on the Linux kernel) is reclaimed - // BEFORE the next language's store-load — instead of leaving V8 to collect - // it lazily under the next pass's allocation pressure (which, at a cap >= - // RAM, degrades into swap-thrash). Collects only dead objects: the live - // cross-file index of the next pass is untouched. The pre/post probe - // confirms whether old-space fragmentation defeats the reclaim. - logHeapProbe('lang-release-pre-gc', `lang=${lang}`); - forceGc(); - logHeapProbe('lang-release-post-gc', `lang=${lang}`); - logHeapProbe('scope-lang-end', `lang=${lang} filesProcessed=${stats.filesProcessed}`); - - processedScopeFiles += langFileCount; - anyRan = true; - functionSummaries.push(...stats.functionSummaries); - totalFiles += stats.filesProcessed; - totalImports += stats.importsEmitted; - totalRefs += stats.referenceEdgesEmitted; - perLanguage.set(lang, { - filesProcessed: stats.filesProcessed, - importsEmitted: stats.importsEmitted, - referenceEdgesEmitted: stats.referenceEdgesEmitted, - }); - - if (isDev) { - logger.info( - `[scope-resolution:${lang}] ${stats.filesProcessed} files → ${stats.importsEmitted} IMPORTS + ${stats.referenceEdgesEmitted} reference edges (${stats.resolve.unresolved} unresolved sites, ${stats.referenceSkipped} skipped)`, + logger.warn( + '[scope-resolution] streaming PDG emit requested but no storage path is ' + + 'available; falling back to in-memory whole-graph emit', ); } } + // Cross-pass per-file dedup set for the streaming sink (#2202): one set + // shared across every language pass so a file emitted in two passes (e.g. a + // `.ts` module pulled into the Vue context pass) streams its PDG layer once. + // Only created when streaming — the in-memory-graph path dedups via its Map. + const pdgEmittedFiles = pdgEmitSink !== undefined ? new Set() : undefined; + + // Stream the PDG layer with guaranteed writer cleanup: a throw escaping the + // per-language loop (outside run.ts's per-file try/catch — e.g. from + // finalize/propagate/a provider hook) must still release the sink's file + // descriptors. finalize() runs on the success path; the finally closes the + // sink only when finalize did not (idempotent via the sink's `finalized`). + let pdgEmitManifest: PdgEmitManifest | undefined; + let pdgSinkSettled = false; + try { + for (const [lang, provider] of SCOPE_RESOLVERS) { + // Standalone providers (COBOL, JCL) don't emit graph edges yet + // through the scope-resolution path. This is the canonical guard: + // runScopeResolution is never called for standalone providers, which + // keeps cobolPhase as the sole IMPORTS edge producer. Keep this guard + // in sync with any additional standalone providers added to + // SCOPE_RESOLVERS. + if (provider.languageProvider.parseStrategy === 'standalone') continue; + + const primaryLangFiles = filesByLang.get(lang) ?? []; + if (primaryLangFiles.length === 0) continue; + const primaryFilePaths = primaryLangFiles.map((f) => f.path); + + // Load per-language import-resolution config (tsconfig paths, + // composer.json autoload, go.mod, ...). One I/O round trip per + // workspace pass — cached implicitly by the result handed to + // every `resolveImportTarget` call below. + const resolutionConfig = + provider.loadResolutionConfig !== undefined + ? await provider.loadResolutionConfig(ctx.repoPath) + : undefined; + + // Some languages (e.g. Vue) expand their file universe beyond the + // primary-language files via the `collectScopeContextPaths` hook. + // The hook receives raw source contents of the primary files so it + // can trace import closures without a second tree-sitter parse. + // + // To avoid reading primary files twice (once for the hook, once for + // the resolution pass), we read them upfront and merge with the + // extra context paths the hook may add. + // Stream this language's pre-built ParsedFiles in from the disk store + // FIRST (huge-repo path). Doing it before reading source lets us skip + // loading content for files the store already covers — for a provider + // with no content-consuming hook that source is pure dead weight once + // extraction is served from the store (~1.5 GB on the kernel's C pass). + // Merged into `preExtractedByPath`; the per-language release block below + // evicts these again before the next language, so only one language's + // ParsedFiles are resident at a time. + const loadStoreFor = async (paths: ReadonlySet): Promise => { + if (!parsedFileStorePath) return; + const fromDisk = await loadParsedFilesForPaths(parsedFileStorePath, paths); + for (const [fp, pf] of fromDisk) preExtractedByPath.set(fp, pf); + }; + + // A provider that feeds source text into a post-extract hook + // (populateWorkspaceOwners / populateNamespaceSiblings / + // populateRangeBindings / emitPostResolutionEdges) needs content for ALL + // its files; one without those hooks only needs content for files the + // store does NOT cover (fresh-extract fallback). Keep this in sync with + // the getFileContents() call-sites in run.ts. + const providerNeedsAllContent = + provider.populateWorkspaceOwners !== undefined || + provider.populateNamespaceSiblings !== undefined || + provider.populateRangeBindings !== undefined || + provider.emitPostResolutionEdges !== undefined; + + let scopeFilePaths: Set; + let contents: Map; + if (provider.collectScopeContextPaths !== undefined) { + // Context-expanding providers (e.g. Vue) need every primary file's + // source up front for the closure hook, so load it all. + const entryFileContents = await readFileContents(ctx.repoPath, primaryFilePaths); + scopeFilePaths = provider.collectScopeContextPaths({ + primaryFilePaths, + preExtractedByPath, + entryFileContents, + allScannedPaths, + resolutionConfig, + }); + // Read only the extra context files (TS/JS etc.) not already loaded. + const extraPaths = [...scopeFilePaths].filter((p) => !entryFileContents.has(p)); + const extraContents = await readFileContents(ctx.repoPath, extraPaths); + contents = new Map([...entryFileContents, ...extraContents]); + await loadStoreFor(scopeFilePaths); + } else { + scopeFilePaths = new Set(primaryFilePaths); + await loadStoreFor(scopeFilePaths); + const pathsToRead = providerNeedsAllContent + ? primaryFilePaths + : primaryFilePaths.filter((p) => !preExtractedByPath.has(p)); + contents = await readFileContents(ctx.repoPath, pathsToRead); + } + const filePaths = [...scopeFilePaths]; + const files: { path: string; content: string }[] = []; + for (const fp of filePaths) { + const content = contents.get(fp); + if (content !== undefined) { + files.push({ path: fp, content }); + } else if (preExtractedByPath.has(fp)) { + // Store covers extraction for this file and we deliberately skipped + // reading its source; the empty string is never consumed (the + // extract loop uses the pre-extracted ParsedFile and this provider + // has no content hook). + files.push({ path: fp, content: '' }); + } + // else: uncovered AND unreadable → skip (unchanged from prior behavior). + } + + const langFileCount = files.length; + logHeapProbe( + 'scope-lang-start', + `lang=${lang} files=${langFileCount} contentsLoaded=${contents.size}`, + ); + const langLabel = lang.charAt(0).toUpperCase() + lang.slice(1); + currentLangIdx++; + const langTag = + totalScopeLangs > 1 ? `${langLabel} [${currentLangIdx}/${totalScopeLangs}]` : langLabel; + + if (totalScopeFiles > 0) { + const pct = + SCOPE_PCT_START + Math.round((processedScopeFiles / totalScopeFiles) * SCOPE_PCT_RANGE); + ctx.onProgress({ + phase: 'scopeResolution', + percent: pct, + message: 'Resolving types', + detail: `${langTag}, ${langFileCount.toLocaleString()} files`, + }); + } + + const stats = runScopeResolution( + { + graph: ctx.graph, + model, + files, + resolutionConfig, + prebuiltNodeLookup: sharedNodeLookup, + prebuiltFunctionNodeIndex: sharedFnNodeIndex, + preExtractedParsedFiles: preExtractedByPath, + scopeIndexStorePath: parsedFileStorePath, + // CFG/PDG emission (#2081 M1) — opt-in; off ⇒ byte-identical graph. + pdg: ctx.options?.pdg === true, + pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction, + pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction, + pdgMaxCdgEdgesPerFunction: ctx.options?.pdgMaxCdgEdgesPerFunction, + pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction, + pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops, + // Streaming PDG-emit sink (#2202) — undefined ⇒ emit to the in-memory graph. + pdgEmitSink, + // Cross-pass per-file dedup set (#2202) — undefined when not streaming. + pdgEmittedFiles, + recordResolutionOutcome: (outcome) => { + resolutionOutcomes.push(outcome); + }, + onWarn: (msg) => { + if (isSemanticModelValidatorEnabled()) { + logger.warn(`[scope-resolution:${lang}] ${msg}`); + } + }, + onProgress: + totalScopeFiles > 0 + ? (subPhase: ScopeResolutionSubPhase, current, total) => { + let langRatio: number; + switch (subPhase) { + case 'extracting': + langRatio = total > 0 ? (current / total) * 0.5 : 0; + break; + case 'analyzing types': + langRatio = 0.5; + break; + case 'resolving references': + langRatio = 0.7; + break; + case 'linking symbols': + langRatio = 0.85; + break; + default: { + const _exhaustive: never = subPhase; + langRatio = 0.85; + } + } + const overallRatio = Math.min( + 1, + (processedScopeFiles + langRatio * langFileCount) / totalScopeFiles, + ); + const pct = SCOPE_PCT_START + Math.round(overallRatio * SCOPE_PCT_RANGE); + ctx.onProgress({ + phase: 'scopeResolution', + percent: pct, + message: 'Resolving types', + detail: + subPhase === 'extracting' + ? `${langTag} — extracting ${current.toLocaleString()}/${total.toLocaleString()} files` + : `${langTag} — ${subPhase}`, + }); + } + : undefined, + }, + provider, + ); + + // Release file contents and pre-extracted entries after each language + // to reduce memory pressure. For large codebases (16K+ PHP files), + // holding all source code simultaneously with scope trees causes OOM. + // See: https://github.com/abhigyanpatwari/GitNexus/issues/1741 + // + // Use `filePaths` (not `primaryFilePaths`) so that any context files + // added by `collectScopeContextPaths` (e.g. TS/JS files pulled in for + // Vue cross-file resolution) are also evicted and not held until GC. + files.length = 0; + contents.clear(); + for (const fp of filePaths) { + preExtractedByPath.delete(fp); + } + // This language's ParsedFiles are now unreachable (runScopeResolution has + // returned and the Map entries are deleted). Force a GC HERE so a heavy + // language's ~17-20GB set (e.g. C/C++ on the Linux kernel) is reclaimed + // BEFORE the next language's store-load — instead of leaving V8 to collect + // it lazily under the next pass's allocation pressure (which, at a cap >= + // RAM, degrades into swap-thrash). Collects only dead objects: the live + // cross-file index of the next pass is untouched. The pre/post probe + // confirms whether old-space fragmentation defeats the reclaim. + logHeapProbe('lang-release-pre-gc', `lang=${lang}`); + forceGc(); + logHeapProbe('lang-release-post-gc', `lang=${lang}`); + logHeapProbe('scope-lang-end', `lang=${lang} filesProcessed=${stats.filesProcessed}`); + + processedScopeFiles += langFileCount; + anyRan = true; + functionSummaries.push(...stats.functionSummaries); + totalFiles += stats.filesProcessed; + totalImports += stats.importsEmitted; + totalRefs += stats.referenceEdgesEmitted; + perLanguage.set(lang, { + filesProcessed: stats.filesProcessed, + importsEmitted: stats.importsEmitted, + referenceEdgesEmitted: stats.referenceEdgesEmitted, + }); + + if (isDev) { + logger.info( + `[scope-resolution:${lang}] ${stats.filesProcessed} files → ${stats.importsEmitted} IMPORTS + ${stats.referenceEdgesEmitted} reference edges (${stats.resolve.unresolved} unresolved sites, ${stats.referenceSkipped} skipped)`, + ); + } + } + + // Finalize the streaming PDG sink (#2202) once after the last language: + // flush + close its CSV writers and capture the COPY manifest. forceGc at + // the boundary reclaims transient write buffers (mirrors the per-language + // release below). + pdgEmitManifest = pdgEmitSink?.finalize(); + pdgSinkSettled = true; + if (pdgEmitSink !== undefined) forceGc(); + } finally { + // Release fds if a throw skipped finalize (idempotent with finalize()). + if (pdgEmitSink !== undefined && !pdgSinkSettled) pdgEmitSink.close(); + } if (totalScopeFiles > 0 && anyRan) { ctx.onProgress({ @@ -494,7 +559,10 @@ export const scopeResolutionPhase: PipelinePhase = { } } - if (!anyRan) return NOOP_OUTPUT; + // Even when no language ran, surface a finalized manifest (its CSVs are on + // disk) so loadGraphToLbug COPYs them rather than orphaning them — empty in + // the no-files case, harmless. + if (!anyRan) return pdgEmitManifest ? { ...NOOP_OUTPUT, pdgEmitManifest } : NOOP_OUTPUT; return { ran: true, @@ -504,6 +572,7 @@ export const scopeResolutionPhase: PipelinePhase = { resolutionOutcomes, perLanguage, functionSummaries, + pdgEmitManifest, }; }, }; diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts index aa30c2416..7b573e0a6 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts @@ -302,6 +302,30 @@ interface RunScopeResolutionInput { * `reason`; consumed by the U4 taint emit step). `undefined` ⇒ * `DEFAULT_PDG_MAX_TAINT_HOPS` (32); `0` ⇒ no cap. */ readonly pdgMaxTaintHops?: number; + /** + * Streaming PDG-emit sink (#2202). When present (streaming on, full rebuild), + * the `--pdg` emit routes BasicBlock nodes + intra-file PDG edges to THIS + * graph-shaped target instead of the in-memory `graph`, so the bulky PDG + * layer never accumulates in memory (peak RSS O(chunk)). Typed as a plain + * `KnowledgeGraph` so this module stays decoupled from the persistence layer; + * the caller (the scope-resolution phase) owns its lifecycle and finalizes it + * after the last language. Absent ⇒ the emit writes to `graph` as before + * (byte-identical default). + */ + readonly pdgEmitSink?: KnowledgeGraph; + /** + * Cross-pass per-file dedup set for streaming PDG emit (#2202). Shared across + * every language pass (owned by the scope-resolution phase). A file imported + * by more than one language (e.g. a `.ts` module pulled into the Vue context + * pass) is PDG-emitted in each pass over the same `cfgSideChannel`, producing + * identical ids; the in-memory graph dedups that by id, but the streaming sink + * is dedup-free (to stay O(write buffer), not O(total ids)). So when present + * (streaming on), the emit loop skips a file whose PDG already streamed and + * records the rest — keeping the streamed set byte-identical to the + * Map-deduped whole-graph emit, for any language-pass order. Absent ⇒ no skip + * (the graph Map dedups), so the default path is unchanged. + */ + readonly pdgEmittedFiles?: Set; /** * Optional graph-node lookup built ONCE by the caller and shared across * every language pass. `buildGraphNodeLookup` scans the whole graph and is @@ -769,6 +793,11 @@ export function runScopeResolution( // can bracket it. Printed as the PROF `taint=` segment. let taintMs = 0; if (input.pdg === true) { + // Streaming target (#2202): when a sink is provided, BasicBlock nodes + + // intra-file PDG edges are routed to CSV-on-disk through it instead of + // accumulating in `graph`. The function-node index below is still built + // from the real `graph` (Function/Method nodes live there, never the sink). + const pdgTarget: KnowledgeGraph = input.pdgEmitSink ?? graph; let cfgBlocks = 0; let cfgEdges = 0; let cfgDroppedEdges = 0; @@ -835,6 +864,15 @@ export function runScopeResolution( // shard that slipped the version gate) must skip emission, not throw a // TypeError mid-graph-build and abort scope-resolution for the language. if (!Array.isArray(cfgs) || cfgs.length === 0) continue; + // Cross-pass per-file dedup (#2202): when streaming, a file whose PDG + // already streamed in a prior language pass (e.g. a `.ts` module pulled + // into the Vue context pass) would re-emit identical ids from the same + // cfgSideChannel — the dedup-free streaming sink would double the rows. + // Skip it here; the in-memory-graph path needs no skip (its Map dedups). + if (input.pdgEmittedFiles !== undefined) { + if (input.pdgEmittedFiles.has(pf.filePath)) continue; + input.pdgEmittedFiles.add(pf.filePath); + } try { // Per-element emit-safety filter (mirrors the parsedfile-store // reviver's POLICY: valid elements in a mixed array still emit; junk @@ -854,7 +892,7 @@ export function runScopeResolution( } if (wellFormed.length === 0) continue; const emitted = emitFileCfgs( - graph, + pdgTarget, wellFormed, input.pdgMaxEdgesPerFunction ?? DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, // Log cap-overflow drops UNCONDITIONALLY (not via input.onWarn, which is @@ -873,7 +911,7 @@ export function runScopeResolution( // PROF-gated like every other checkpoint here (zero cost when off). const t0 = PROF ? performance.now() : 0; const rd = emitFileReachingDefs( - graph, + pdgTarget, wellFormed, input.pdgMaxReachingDefEdgesPerFunction ?? DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, @@ -892,7 +930,7 @@ export function runScopeResolution( // persisted and its time folds into the `pdg=` PROF segment next to RD. const tCdg = PROF ? performance.now() : 0; const cdg = emitFileCdg( - graph, + pdgTarget, wellFormed, input.pdgMaxCdgEdgesPerFunction ?? DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, (message) => logger.warn(message), // unconditional — R6, no silent truncation @@ -909,7 +947,7 @@ export function runScopeResolution( if (taintSpec !== undefined) { const t1 = PROF ? performance.now() : 0; const taint = emitFileTaint( - graph, + pdgTarget, wellFormed, pf.parsedImports, taintSpec, diff --git a/gitnexus/src/core/ingestion/utils/env.ts b/gitnexus/src/core/ingestion/utils/env.ts index 5beeb818f..70b3baa3f 100644 --- a/gitnexus/src/core/ingestion/utils/env.ts +++ b/gitnexus/src/core/ingestion/utils/env.ts @@ -28,6 +28,18 @@ export const parseTruthyEnv = (raw: string | undefined): boolean => { return value === '1' || value === 'true' || value === 'yes'; }; +/** + * Parse a positive-integer env-var value. Returns the integer when `raw` is a + * finite integer `> 0`; otherwise (`undefined`, empty, non-numeric, `0`, or + * negative) returns `undefined` so the caller falls back to its default. Used + * for numeric tuning knobs like `GITNEXUS_PDG_EMIT_CHUNK_SIZE` (#2202). + */ +export const parsePositiveIntEnv = (raw: string | undefined): number | undefined => { + if (raw === undefined) return undefined; + const n = Number(raw.trim()); + return Number.isInteger(n) && n > 0 ? n : undefined; +}; + /** * Whether scope-resolution dev validators (e.g. `validateBindingsImmutability`) * should run AND emit warnings. Off by default in CLI runs to avoid silent diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index c7d8413b5..af756c74f 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -17,7 +17,8 @@ import { createWriteStream, WriteStream } from 'fs'; import path from 'path'; import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; import { KnowledgeGraph } from '../graph/types.js'; -import { NodeTableName } from './schema.js'; +import { NodeTableName, NODE_TABLES } from './schema.js'; +import { RelPairRouter } from './rel-pair-routing.js'; import { parseTruthyEnv } from '../ingestion/utils/env.js'; /** @@ -182,13 +183,20 @@ class BufferedCSVWriter { this.buffer.push(header); } - addRow(row: string) { + /** + * Buffer a row. Returns a promise ONLY when the buffer crossed FLUSH_EVERY + * and a disk write was issued; otherwise returns `undefined` so the caller + * can skip awaiting (#2203 U3) — avoiding a microtask tick on every buffered + * row (millions at scale). The flush promise still resolves on drain, so + * backpressure is preserved on the rows that actually write. + */ + addRow(row: string): Promise | undefined { this.buffer.push(row); this.rows++; if (this.buffer.length >= FLUSH_EVERY) { return this.flush(); } - return Promise.resolve(); + return undefined; } flush(): Promise { @@ -223,10 +231,51 @@ class BufferedCSVWriter { // STREAMING CSV GENERATION — SINGLE PASS // ============================================================================ +/** Canonical relationship CSV header — shared by the emit pass and the + * `splitRelCsvByLabelPair` differential oracle. */ +export const REL_CSV_HEADER = 'from,to,type,confidence,reason,step'; + +/** Build the escaped CSV row (no trailing newline) for one relationship. + * Single source of the relationship row bytes — used by the emit pass and by + * the byte-identity differential test that feeds the legacy split oracle. */ +export const buildRelRow = (rel: GraphRelationship): string => + [ + escapeCSVField(rel.sourceId), + escapeCSVField(rel.targetId), + escapeCSVField(rel.type), + escapeCSVNumber(rel.confidence, 1.0), + escapeCSVField(rel.reason), + escapeCSVNumber(rel.step, 0), + ].join(','); + +/** Canonical BasicBlock node CSV header — taint/PDG substrate (issue #2080). + * No `name` column; blocks are identified by id + source span. Shared by the + * whole-graph emit pass and the streaming PDG emit sink (issue #2202) so the + * two paths produce byte-identical BasicBlock rows by construction. */ +export const BASICBLOCK_CSV_HEADER = 'id,filePath,startLine,endLine,text'; + +/** Build the escaped CSV row (no trailing newline) for one BasicBlock node. + * Single source of the BasicBlock row bytes — used by `streamAllCSVsToDisk` + * and by the streaming `PdgEmitSink` (issue #2202). */ +export const buildBasicBlockRow = (node: GraphNode): string => + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.filePath || ''), + escapeCSVNumber(node.properties.startLine, -1), + escapeCSVNumber(node.properties.endLine, -1), + escapeCSVField(node.properties.text || ''), + ].join(','); + export interface StreamedCSVResult { nodeFiles: Map; - relCsvPath: string; - relRows: number; + /** pairKey (`From|To`) → per-FROM→TO-label-pair CSV file. */ + relsByPair: Map; + /** Header line shared by every per-pair file. */ + relHeader: string; + /** Edges skipped because an endpoint label is not a valid node table. */ + skippedRels: number; + /** Edges routed to a per-pair file. */ + totalValidRels: number; } /** @@ -253,255 +302,185 @@ export const streamAllCSVsToDisk = async ( const prevMax = process.getMaxListeners(); process.setMaxListeners(prevMax + 40); - const contentCache = new FileContentCache(repoPath); + // try/finally so the listener bump is ALWAYS restored — including the + // rel-routing throw path (#2203 U2) and any node-writer finish() rejection, + // not just the success path (avoids leaking +40 listeners across failed runs + // in long-lived hosts / the test suite). + try { + const contentCache = new FileContentCache(repoPath); - // Create writers for every node type up-front - const fileWriter = new BufferedCSVWriter( - path.join(csvDir, 'file.csv'), - 'id,name,filePath,content', - ); - const folderWriter = new BufferedCSVWriter(path.join(csvDir, 'folder.csv'), 'id,name,filePath'); - const codeElementHeader = 'id,name,filePath,startLine,endLine,isExported,content,description'; - const functionWriter = new BufferedCSVWriter( - path.join(csvDir, 'function.csv'), - codeElementHeader, - ); - const classWriter = new BufferedCSVWriter(path.join(csvDir, 'class.csv'), codeElementHeader); - const interfaceWriter = new BufferedCSVWriter( - path.join(csvDir, 'interface.csv'), - codeElementHeader, - ); - const methodHeader = - 'id,name,filePath,startLine,endLine,isExported,content,description,parameterCount,returnType'; - const methodWriter = new BufferedCSVWriter(path.join(csvDir, 'method.csv'), methodHeader); - const codeElemWriter = new BufferedCSVWriter( - path.join(csvDir, 'codeelement.csv'), - codeElementHeader, - ); - const communityWriter = new BufferedCSVWriter( - path.join(csvDir, 'community.csv'), - 'id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount', - ); - const processWriter = new BufferedCSVWriter( - path.join(csvDir, 'process.csv'), - 'id,label,heuristicLabel,processType,stepCount,communities,entryPointId,terminalId', - ); - - // Section nodes have an extra 'level' column - const sectionWriter = new BufferedCSVWriter( - path.join(csvDir, 'section.csv'), - 'id,name,filePath,startLine,endLine,level,content,description', - ); - - // Route nodes for API endpoint mapping - const routeWriter = new BufferedCSVWriter( - path.join(csvDir, 'route.csv'), - 'id,name,filePath,responseKeys,errorKeys,middleware', - ); - - // Tool nodes for MCP tool definitions - const toolWriter = new BufferedCSVWriter( - path.join(csvDir, 'tool.csv'), - 'id,name,filePath,description', - ); - - // BasicBlock nodes — taint/PDG substrate (issue #2080). No `name` column; - // blocks are identified by id + source span. Emitted by no phase yet. - const basicBlockWriter = new BufferedCSVWriter( - path.join(csvDir, 'basicblock.csv'), - 'id,filePath,startLine,endLine,text', - ); - - // Multi-language node types share the same CSV shape (no isExported column) - const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description'; - const MULTI_LANG_TYPES = [ - 'Struct', - 'Enum', - 'Macro', - 'Typedef', - 'Union', - 'Namespace', - 'Trait', - 'Impl', - 'TypeAlias', - 'Const', - 'Static', - 'Variable', - 'Property', - 'Record', - 'Delegate', - 'Annotation', - 'Constructor', - 'Template', - 'Module', - ] as const; - const propertyHeader = 'id,name,filePath,startLine,endLine,content,description,declaredType'; - const multiLangWriters = new Map(); - for (const t of MULTI_LANG_TYPES) { - multiLangWriters.set( - t, - new BufferedCSVWriter( - path.join(csvDir, `${t.toLowerCase()}.csv`), - t === 'Property' ? propertyHeader : multiLangHeader, - ), + // Create writers for every node type up-front + const fileWriter = new BufferedCSVWriter( + path.join(csvDir, 'file.csv'), + 'id,name,filePath,content', + ); + const folderWriter = new BufferedCSVWriter(path.join(csvDir, 'folder.csv'), 'id,name,filePath'); + const codeElementHeader = 'id,name,filePath,startLine,endLine,isExported,content,description'; + const functionWriter = new BufferedCSVWriter( + path.join(csvDir, 'function.csv'), + codeElementHeader, + ); + const classWriter = new BufferedCSVWriter(path.join(csvDir, 'class.csv'), codeElementHeader); + const interfaceWriter = new BufferedCSVWriter( + path.join(csvDir, 'interface.csv'), + codeElementHeader, + ); + const methodHeader = + 'id,name,filePath,startLine,endLine,isExported,content,description,parameterCount,returnType'; + const methodWriter = new BufferedCSVWriter(path.join(csvDir, 'method.csv'), methodHeader); + const codeElemWriter = new BufferedCSVWriter( + path.join(csvDir, 'codeelement.csv'), + codeElementHeader, + ); + const communityWriter = new BufferedCSVWriter( + path.join(csvDir, 'community.csv'), + 'id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount', + ); + const processWriter = new BufferedCSVWriter( + path.join(csvDir, 'process.csv'), + 'id,label,heuristicLabel,processType,stepCount,communities,entryPointId,terminalId', ); - } - const codeWriterMap: Record = { - Function: functionWriter, - Class: classWriter, - Interface: interfaceWriter, - CodeElement: codeElemWriter, - }; + // Section nodes have an extra 'level' column + const sectionWriter = new BufferedCSVWriter( + path.join(csvDir, 'section.csv'), + 'id,name,filePath,startLine,endLine,level,content,description', + ); - // Deduplicate all node types — the pipeline can produce duplicate IDs across - // all symbol types (Class, Method, Function, etc.), not just File nodes. - // A single Set covering every label prevents PK violations on COPY. - const seenNodeIds = new Set(); + // Route nodes for API endpoint mapping + const routeWriter = new BufferedCSVWriter( + path.join(csvDir, 'route.csv'), + 'id,name,filePath,responseKeys,errorKeys,middleware', + ); - // --- SINGLE PASS over all nodes --- - for (const node of orderedNodes(graph, sortOutput)) { - if (seenNodeIds.has(node.id)) continue; - seenNodeIds.add(node.id); + // Tool nodes for MCP tool definitions + const toolWriter = new BufferedCSVWriter( + path.join(csvDir, 'tool.csv'), + 'id,name,filePath,description', + ); - switch (node.label) { - case 'File': { - const content = await extractContent(node, contentCache); - await fileWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVField(content), - ].join(','), - ); - break; - } - case 'Folder': - await folderWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - ].join(','), - ); - break; - case 'Community': { - const keywords = node.properties.keywords || []; - const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/\\/g, '\\\\').replace(/'/g, "''").replace(/,/g, '\\,')}'`).join(',')}]`; - await communityWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.heuristicLabel || ''), - keywordsStr, - escapeCSVField(node.properties.description || ''), - escapeCSVField(node.properties.enrichedBy || 'heuristic'), - escapeCSVNumber(node.properties.cohesion, 0), - escapeCSVNumber(node.properties.symbolCount, 0), - ].join(','), - ); - break; - } - case 'Process': { - const communities = node.properties.communities || []; - const communitiesStr = `[${communities.map((c: string) => `'${c.replace(/'/g, "''")}'`).join(',')}]`; - await processWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.heuristicLabel || ''), - escapeCSVField(node.properties.processType || ''), - escapeCSVNumber(node.properties.stepCount, 0), - escapeCSVField(communitiesStr), - escapeCSVField(node.properties.entryPointId || ''), - escapeCSVField(node.properties.terminalId || ''), - ].join(','), - ); - break; - } - case 'Method': { - const content = await extractContent(node, contentCache); - await methodWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVNumber(node.properties.startLine, -1), - escapeCSVNumber(node.properties.endLine, -1), - node.properties.isExported ? 'true' : 'false', - escapeCSVField(content), - escapeCSVField(node.properties.description || ''), - escapeCSVNumber(node.properties.parameterCount, 0), - escapeCSVField(node.properties.returnType || ''), - ].join(','), - ); - break; - } - case 'Section': { - const content = await extractContent(node, contentCache); - await sectionWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVNumber(node.properties.startLine, -1), - escapeCSVNumber(node.properties.endLine, -1), - escapeCSVNumber(node.properties.level, 1), - escapeCSVField(content), - escapeCSVField(node.properties.description || ''), - ].join(','), - ); - break; - } - case 'Route': { - const responseKeys = node.properties.responseKeys || []; - // LadybugDB array literal inside a quoted CSV field: escapeCSVField wraps in "..." - // and the array uses single-quoted elements - const keysStr = `[${responseKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`; - const errorKeys = node.properties.errorKeys || []; - const errorKeysStr = `[${errorKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`; - const middleware = node.properties.middleware || []; - const middlewareStr = `[${middleware.map((m: string) => `'${m.replace(/'/g, "''")}'`).join(',')}]`; - await routeWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVField(keysStr), - escapeCSVField(errorKeysStr), - escapeCSVField(middlewareStr), - ].join(','), - ); - break; - } - case 'Tool': - await toolWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVField(node.properties.description || ''), - ].join(','), - ); - break; - case 'BasicBlock': - await basicBlockWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.filePath || ''), - escapeCSVNumber(node.properties.startLine, -1), - escapeCSVNumber(node.properties.endLine, -1), - escapeCSVField(node.properties.text || ''), - ].join(','), - ); - break; - default: { - // Code element nodes (Function, Class, Interface, CodeElement) - const writer = codeWriterMap[node.label]; - if (writer) { + // BasicBlock nodes — taint/PDG substrate (issue #2080). No `name` column; + // blocks are identified by id + source span. Emitted by no phase yet. + const basicBlockWriter = new BufferedCSVWriter( + path.join(csvDir, 'basicblock.csv'), + BASICBLOCK_CSV_HEADER, + ); + + // Multi-language node types share the same CSV shape (no isExported column) + const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description'; + const MULTI_LANG_TYPES = [ + 'Struct', + 'Enum', + 'Macro', + 'Typedef', + 'Union', + 'Namespace', + 'Trait', + 'Impl', + 'TypeAlias', + 'Const', + 'Static', + 'Variable', + 'Property', + 'Record', + 'Delegate', + 'Annotation', + 'Constructor', + 'Template', + 'Module', + ] as const; + const propertyHeader = 'id,name,filePath,startLine,endLine,content,description,declaredType'; + const multiLangWriters = new Map(); + for (const t of MULTI_LANG_TYPES) { + multiLangWriters.set( + t, + new BufferedCSVWriter( + path.join(csvDir, `${t.toLowerCase()}.csv`), + t === 'Property' ? propertyHeader : multiLangHeader, + ), + ); + } + + const codeWriterMap: Record = { + Function: functionWriter, + Class: classWriter, + Interface: interfaceWriter, + CodeElement: codeElemWriter, + }; + + // Deduplicate all node types — the pipeline can produce duplicate IDs across + // all symbol types (Class, Method, Function, etc.), not just File nodes. + // A single Set covering every label prevents PK violations on COPY. + const seenNodeIds = new Set(); + + // --- SINGLE PASS over all nodes --- + for (const node of orderedNodes(graph, sortOutput)) { + if (seenNodeIds.has(node.id)) continue; + seenNodeIds.add(node.id); + + // addRow returns a promise only when it flushes; awaiting it once after the + // switch (instead of `await`-ing every addRow) skips a per-row microtask + // tick on the ~FLUSH_EVERY-1 buffered rows between flushes (#2203 U3). + let pending: Promise | undefined; + switch (node.label) { + case 'File': { const content = await extractContent(node, contentCache); - await writer.addRow( + pending = fileWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVField(content), + ].join(','), + ); + break; + } + case 'Folder': + pending = folderWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + ].join(','), + ); + break; + case 'Community': { + const keywords = node.properties.keywords || []; + const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/\\/g, '\\\\').replace(/'/g, "''").replace(/,/g, '\\,')}'`).join(',')}]`; + pending = communityWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.heuristicLabel || ''), + keywordsStr, + escapeCSVField(node.properties.description || ''), + escapeCSVField(node.properties.enrichedBy || 'heuristic'), + escapeCSVNumber(node.properties.cohesion, 0), + escapeCSVNumber(node.properties.symbolCount, 0), + ].join(','), + ); + break; + } + case 'Process': { + const communities = node.properties.communities || []; + const communitiesStr = `[${communities.map((c: string) => `'${c.replace(/'/g, "''")}'`).join(',')}]`; + pending = processWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.heuristicLabel || ''), + escapeCSVField(node.properties.processType || ''), + escapeCSVNumber(node.properties.stepCount, 0), + escapeCSVField(communitiesStr), + escapeCSVField(node.properties.entryPointId || ''), + escapeCSVField(node.properties.terminalId || ''), + ].join(','), + ); + break; + } + case 'Method': { + const content = await extractContent(node, contentCache); + pending = methodWriter.addRow( [ escapeCSVField(node.id), escapeCSVField(node.properties.name || ''), @@ -511,101 +490,191 @@ export const streamAllCSVsToDisk = async ( node.properties.isExported ? 'true' : 'false', escapeCSVField(content), escapeCSVField(node.properties.description || ''), + escapeCSVNumber(node.properties.parameterCount, 0), + escapeCSVField(node.properties.returnType || ''), ].join(','), ); - } else { - // Multi-language node types (Struct, Impl, Trait, Macro, etc.) - const mlWriter = multiLangWriters.get(node.label); - if (mlWriter) { + break; + } + case 'Section': { + const content = await extractContent(node, contentCache); + pending = sectionWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVNumber(node.properties.startLine, -1), + escapeCSVNumber(node.properties.endLine, -1), + escapeCSVNumber(node.properties.level, 1), + escapeCSVField(content), + escapeCSVField(node.properties.description || ''), + ].join(','), + ); + break; + } + case 'Route': { + const responseKeys = node.properties.responseKeys || []; + // LadybugDB array literal inside a quoted CSV field: escapeCSVField wraps in "..." + // and the array uses single-quoted elements + const keysStr = `[${responseKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`; + const errorKeys = node.properties.errorKeys || []; + const errorKeysStr = `[${errorKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`; + const middleware = node.properties.middleware || []; + const middlewareStr = `[${middleware.map((m: string) => `'${m.replace(/'/g, "''")}'`).join(',')}]`; + pending = routeWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVField(keysStr), + escapeCSVField(errorKeysStr), + escapeCSVField(middlewareStr), + ].join(','), + ); + break; + } + case 'Tool': + pending = toolWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVField(node.properties.description || ''), + ].join(','), + ); + break; + case 'BasicBlock': + pending = basicBlockWriter.addRow(buildBasicBlockRow(node)); + break; + default: { + // Code element nodes (Function, Class, Interface, CodeElement) + const writer = codeWriterMap[node.label]; + if (writer) { const content = await extractContent(node, contentCache); - await mlWriter.addRow( + pending = writer.addRow( [ escapeCSVField(node.id), escapeCSVField(node.properties.name || ''), escapeCSVField(node.properties.filePath || ''), escapeCSVNumber(node.properties.startLine, -1), escapeCSVNumber(node.properties.endLine, -1), + node.properties.isExported ? 'true' : 'false', escapeCSVField(content), escapeCSVField(node.properties.description || ''), - ...(node.label === 'Property' - ? [escapeCSVField(node.properties.declaredType || '')] - : []), ].join(','), ); + } else { + // Multi-language node types (Struct, Impl, Trait, Macro, etc.) + const mlWriter = multiLangWriters.get(node.label); + if (mlWriter) { + const content = await extractContent(node, contentCache); + pending = mlWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVNumber(node.properties.startLine, -1), + escapeCSVNumber(node.properties.endLine, -1), + escapeCSVField(content), + escapeCSVField(node.properties.description || ''), + ...(node.label === 'Property' + ? [escapeCSVField(node.properties.declaredType || '')] + : []), + ].join(','), + ); + } else { + // Unknown label: not in codeWriterMap or multiLangWriters, so there + // is no CSV table for it and it is intentionally NOT persisted — + // `pending` stays undefined, so the loop awaits nothing. Made + // explicit so a future node type isn't silently dropped here: wire + // it into one of the writer maps above (or this branch). + } } + break; } - break; + } + if (pending) await pending; + } + + // Finish all node writers + const allWriters = [ + fileWriter, + folderWriter, + functionWriter, + classWriter, + interfaceWriter, + methodWriter, + codeElemWriter, + communityWriter, + processWriter, + sectionWriter, + routeWriter, + toolWriter, + basicBlockWriter, + ...multiLangWriters.values(), + ]; + await Promise.all(allWriters.map((w) => w.finish())); + + // --- Stream relationships directly to per-FROM→TO-label-pair files --- + // (#2203 U2) Route every edge to its pair file in this single pass. The old + // monolithic relations.csv — and its line-by-line re-read + per-edge regex + // re-split in loadGraphToLbug — are gone, so the ~1M-edge set is written and + // read once instead of twice. The router applies the SAME label-derivation + + // validTables filter as the legacy splitRelCsvByLabelPair, so the per-pair + // files are byte-identical (asserted by the differential test). + const relRouter = new RelPairRouter(csvDir, REL_CSV_HEADER, new Set(NODE_TABLES)); + try { + for (const rel of orderedRelationships(graph, sortOutput)) { + const pending = relRouter.route(rel.sourceId, rel.targetId, buildRelRow(rel)); + if (pending) await pending; + } + await relRouter.close(); + } catch (err) { + relRouter.destroy(); + // Rethrow the real stream error (EMFILE / disk-full) rather than the generic + // AbortError a pending drain-await rejects with — mirrors the retained + // splitRelCsvByLabelPair's `throw streamError ?? err`. + throw relRouter.lastError ?? err; + } + + // Build result map — only include tables that have rows + const nodeFiles = new Map(); + const tableMap: [NodeTableName, BufferedCSVWriter][] = [ + ['File', fileWriter], + ['Folder', folderWriter], + ['Function', functionWriter], + ['Class', classWriter], + ['Interface', interfaceWriter], + ['Method', methodWriter], + ['CodeElement', codeElemWriter], + ['Community', communityWriter], + ['Process', processWriter], + ['Section' as NodeTableName, sectionWriter], + ['Route' as NodeTableName, routeWriter], + ['Tool' as NodeTableName, toolWriter], + ['BasicBlock' as NodeTableName, basicBlockWriter], + ...Array.from(multiLangWriters.entries()).map( + ([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter], + ), + ]; + for (const [name, writer] of tableMap) { + if (writer.rows > 0) { + nodeFiles.set(name, { + csvPath: path.join(csvDir, `${name.toLowerCase()}.csv`), + rows: writer.rows, + }); } } + + return { + nodeFiles, + relsByPair: relRouter.byPair, + relHeader: REL_CSV_HEADER, + skippedRels: relRouter.skipped, + totalValidRels: relRouter.total, + }; + } finally { + // Restore original process listener limit on every path (success or throw). + process.setMaxListeners(prevMax); } - - // Finish all node writers - const allWriters = [ - fileWriter, - folderWriter, - functionWriter, - classWriter, - interfaceWriter, - methodWriter, - codeElemWriter, - communityWriter, - processWriter, - sectionWriter, - routeWriter, - toolWriter, - basicBlockWriter, - ...multiLangWriters.values(), - ]; - await Promise.all(allWriters.map((w) => w.finish())); - - // --- Stream relationship CSV --- - const relCsvPath = path.join(csvDir, 'relations.csv'); - const relWriter = new BufferedCSVWriter(relCsvPath, 'from,to,type,confidence,reason,step'); - for (const rel of orderedRelationships(graph, sortOutput)) { - await relWriter.addRow( - [ - escapeCSVField(rel.sourceId), - escapeCSVField(rel.targetId), - escapeCSVField(rel.type), - escapeCSVNumber(rel.confidence, 1.0), - escapeCSVField(rel.reason), - escapeCSVNumber((rel as any).step, 0), - ].join(','), - ); - } - await relWriter.finish(); - - // Build result map — only include tables that have rows - const nodeFiles = new Map(); - const tableMap: [NodeTableName, BufferedCSVWriter][] = [ - ['File', fileWriter], - ['Folder', folderWriter], - ['Function', functionWriter], - ['Class', classWriter], - ['Interface', interfaceWriter], - ['Method', methodWriter], - ['CodeElement', codeElemWriter], - ['Community', communityWriter], - ['Process', processWriter], - ['Section' as NodeTableName, sectionWriter], - ['Route' as NodeTableName, routeWriter], - ['Tool' as NodeTableName, toolWriter], - ['BasicBlock' as NodeTableName, basicBlockWriter], - ...Array.from(multiLangWriters.entries()).map( - ([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter], - ), - ]; - for (const [name, writer] of tableMap) { - if (writer.rows > 0) { - nodeFiles.set(name, { - csvPath: path.join(csvDir, `${name.toLowerCase()}.csv`), - rows: writer.rows, - }); - } - } - - // Restore original process listener limit - process.setMaxListeners(prevMax); - - return { nodeFiles, relCsvPath, relRows: relWriter.rows }; }; diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index 158d324e8..1b0c08738 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -4,8 +4,6 @@ import { createInterface } from 'readline'; import { once } from 'events'; import { finished } from 'stream/promises'; import path from 'path'; -import os from 'os'; -import crypto from 'crypto'; import lbug from '@ladybugdb/core'; import { closeQueryResults } from './query-result-utils.js'; import { KnowledgeGraph } from '../graph/types.js'; @@ -19,6 +17,8 @@ import { NodeTableName, } from './schema.js'; import { streamAllCSVsToDisk } from './csv-generator.js'; +import type { PdgEmitManifest } from './pdg-emit-sink.js'; +import { getNodeLabel as deriveNodeLabel, type WriteStreamFactory } from './rel-pair-routing.js'; import type { CachedEmbedding } from '../embeddings/types.js'; import { extensionManager, type ExtensionEnsureOptions } from './extension-loader.js'; import { @@ -28,6 +28,7 @@ import { isWalCorruptionError, openLbugConnection, toNativeSafePath, + resolveNativeSafeStorageDir, WAL_RECOVERY_SUGGESTION, waitForWindowsHandleRelease, type LbugConnectionHandle, @@ -48,9 +49,9 @@ import { logger } from '../logger.js'; // --------------------------------------------------------------------------- // Relationship CSV splitting — extracted for testability (PR #818) // --------------------------------------------------------------------------- - -/** Factory for creating WriteStreams — injectable for testing. */ -export type WriteStreamFactory = (filePath: string) => import('fs').WriteStream; +// WriteStreamFactory is imported above from rel-pair-routing.ts (its canonical +// home) for splitRelCsvByLabelPair's signature; no external code imports it from +// here, so it is not re-exported. /** Result of splitting the relationship CSV into per-label-pair files. */ export interface RelCsvSplitResult { @@ -64,6 +65,15 @@ export interface RelCsvSplitResult { /** * Split a relationship CSV into per-label-pair files on disk. * + * @internal RETAINED AS A DIFFERENTIAL ORACLE. As of #2203 U2, production emit + * routes relationships to per-pair files directly during the single pass (see + * RelPairRouter in `rel-pair-routing.ts`), so this function has NO production + * callers — it is kept ONLY so the byte-identity test in + * `test/integration/csv-pipeline.test.ts` ("direct per-pair emit matches the + * split oracle") can diff the direct-emit output against this proven path. Do + * NOT delete it as dead code without also removing that test and accepting the + * loss of the byte-identity guard (and likewise `test/unit/rel-csv-split.test.ts`). + * * Streams the CSV line-by-line, routing each relationship to a file named * `rel_{fromLabel}_{toLabel}.csv`. Handles backpressure correctly: only one * drain listener per stream at a time, and readline resumes only when ALL @@ -871,6 +881,15 @@ export const loadGraphToLbug = async ( repoPath: string, storagePath: string, onProgress?: LbugProgressCallback, + /** + * Streamed PDG-emit manifest (#2202). When present (streaming was on, full + * rebuild), the BasicBlock node CSV + per-pair PDG-edge CSVs it points at + * were already flushed to disk during the emit loop; they are merged into the + * COPY plan below so they load alongside the structural CSVs. When streaming + * was on the in-memory `graph` holds zero BasicBlocks, so `streamAllCSVsToDisk` + * emits none — the manifest is the sole source and there is no double-COPY. + */ + pdgEmitManifest?: PdgEmitManifest, ) => { if (!conn) { throw new Error('LadybugDB not initialized. Call initLbug first.'); @@ -878,23 +897,57 @@ export const loadGraphToLbug = async ( const log = onProgress || (() => {}); - let csvDir: string; - if (process.platform === 'win32' && /[^\x00-\x7F]/.test(storagePath)) { - const hash = crypto.createHash('sha256').update(storagePath).digest('hex').slice(0, 16); - csvDir = toNativeSafePath(path.join(os.tmpdir(), `gitnexus-csv-${hash}`)); - } else { - csvDir = path.join(storagePath, 'csv'); - } + // ── #2203 persistence-path profiling ────────────────────────────────── + // Mirrors the PROF_SCOPE_RESOLUTION pattern (scope-resolution/pipeline/ + // run.ts): zero-cost when off — process.hrtime.bigint() is only read under + // PROF_LBUG_LOAD=1, and the summary is logged behind the same gate. Fills + // the gap that the DB-persistence path is un-timed today (the analyze + // "emit" number is the scope-resolution emit bucket, not this COPY path). + const PROF = process.env.PROF_LBUG_LOAD === '1'; + const mark = (): bigint => (PROF ? process.hrtime.bigint() : 0n); + const span = (a: bigint, b: bigint): string => (Number(b - a) / 1e6).toFixed(1); + const tStart = mark(); + + const csvDir = resolveNativeSafeStorageDir(storagePath, 'csv'); log('Streaming CSVs to disk...'); const csvResult = await streamAllCSVsToDisk(graph, repoPath, csvDir); + // Merge the streamed PDG-emit CSVs (#2202) into the COPY plan so the + // BasicBlock node table + per-pair PDG edges (CFG / REACHING_DEF / CDG / + // POST_DOMINATE / TAINTED / SANITIZES) load through the SAME node + per-pair + // COPY loops as the structural CSVs. The graph held zero BasicBlocks when + // streaming, so `streamAllCSVsToDisk` produced none of these — the manifest + // is the sole source and there is no double-COPY. Absent ⇒ no-op. + if (pdgEmitManifest) { + for (const [table, meta] of pdgEmitManifest.nodeFiles) { + // A collision means a BasicBlock leaked into the in-memory graph during a + // streamed run (streamAllCSVsToDisk then emitted a structural basicblock.csv). + // That is a streaming-invariant violation — fail loudly rather than + // silently overwrite one CSV with the other and drop its rows (#2202 review #3). + if (csvResult.nodeFiles.has(table)) { + throw new Error( + `Streaming PDG manifest collides with a structural node CSV for "${table}" — ` + + `the in-memory graph should hold zero ${table} nodes when streaming. ` + + `A ${table} node leaked into the graph during a streamed emit.`, + ); + } + csvResult.nodeFiles.set(table, meta); + } + for (const [pairKey, meta] of pdgEmitManifest.relsByPair) { + if (csvResult.relsByPair.has(pairKey)) { + throw new Error( + `Streaming PDG manifest collides with a structural relationship CSV for pair ` + + `"${pairKey}" — a PDG edge leaked into the in-memory graph during a streamed emit.`, + ); + } + csvResult.relsByPair.set(pairKey, meta); + csvResult.totalValidRels += meta.rows; + } + } + const tCsv = mark(); + const validTables = new Set(NODE_TABLES as readonly string[]); - const getNodeLabel = (nodeId: string): string => { - if (nodeId.startsWith('comm_')) return 'Community'; - if (nodeId.startsWith('proc_')) return 'Process'; - return nodeId.split(':')[0]; - }; // Bulk COPY all node CSVs (sequential — LadybugDB allows only one write txn at a time) const nodeFiles = [...csvResult.nodeFiles.entries()]; @@ -924,37 +977,32 @@ export const loadGraphToLbug = async ( } } - // Bulk COPY relationships — split by FROM→TO label pair (LadybugDB requires it) - const { relHeader, relsByPairMeta, pairWriteStreams, skippedRels, totalValidRels } = - await splitRelCsvByLabelPair(csvResult.relCsvPath, csvDir, validTables, getNodeLabel); + const tCopyNodes = mark(); - // Close all per-pair write streams before COPY. `stream/promises.finished` - // resolves on the stream's 'finish' event and rejects on 'error' — replaces - // a hand-rolled promisification with the stdlib primitive. - await Promise.all( - Array.from(pairWriteStreams.values()).map(async (ws) => { - ws.end(); - await finished(ws); - }), - ); + // Bulk COPY relationships. They were already routed to per-FROM→TO-label-pair + // files during the emit pass (#2203 U2) — there is no monolithic relations.csv + // to re-read/re-split here; we COPY each pair file directly. + const { relsByPair, relHeader, skippedRels, totalValidRels } = csvResult; + let tCopyRels = tCopyNodes; + let tFallback = tCopyNodes; const insertedRels = totalValidRels; const warnings: string[] = []; if (insertedRels > 0) { - log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPairMeta.size} types`); + log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPair.size} types`); let pairIdx = 0; let failedPairEdges = 0; const failedPairCsvPaths = new Set(); - for (const [pairKey, { csvPath: pairCsvPath, rows }] of relsByPairMeta) { + for (const [pairKey, { csvPath: pairCsvPath, rows }] of relsByPair) { pairIdx++; const [fromLabel, toLabel] = pairKey.split('|'); const normalizedPath = normalizeCopyPath(pairCsvPath); const copyQuery = `COPY ${REL_TABLE_NAME} FROM "${normalizedPath}" (from="${fromLabel}", to="${toLabel}", HEADER=true, ESCAPE='"', DELIM=',', QUOTE='"', PARALLEL=false, auto_detect=false)`; if (pairIdx % 5 === 0 || rows > 1000) { - log(`Loading edges: ${pairIdx}/${relsByPairMeta.size} types (${fromLabel} -> ${toLabel})`); + log(`Loading edges: ${pairIdx}/${relsByPair.size} types (${fromLabel} -> ${toLabel})`); } try { @@ -980,6 +1028,7 @@ export const loadGraphToLbug = async ( } catch {} } } + tCopyRels = mark(); if (failedPairCsvPaths.size > 0) { log(`Inserting ${failedPairEdges} edges individually (missing schema pairs)`); @@ -999,15 +1048,14 @@ export const loadGraphToLbug = async ( } catch {} } if (allLines.length > 1) { - await fallbackRelationshipInserts(allLines, validTables, getNodeLabel); + await fallbackRelationshipInserts(allLines, validTables, deriveNodeLabel); } } + tFallback = mark(); } - // Cleanup all CSVs - try { - await fs.unlink(csvResult.relCsvPath); - } catch {} + // Cleanup all CSVs (per-pair rel files are unlinked in the COPY loop above; + // the remaining sweep below catches node CSVs + any leftover pair files). for (const [, { csvPath }] of csvResult.nodeFiles) { try { await fs.unlink(csvPath); @@ -1025,6 +1073,18 @@ export const loadGraphToLbug = async ( await fs.rmdir(csvDir); } catch {} + if (PROF) { + const tEnd = mark(); + let totalNodeRows = 0; + for (const [, { rows }] of csvResult.nodeFiles) totalNodeRows += rows; + logger.warn( + `[lbug-load prof] csv-emit=${span(tStart, tCsv)}ms ` + + `copy-nodes=${span(tCsv, tCopyNodes)}ms copy-rels=${span(tCopyNodes, tCopyRels)}ms ` + + `fallback=${span(tCopyRels, tFallback)}ms total=${span(tStart, tEnd)}ms ` + + `(${totalNodeRows} nodes, ${insertedRels} rels)`, + ); + } + return { success: true, insertedRels, skippedRels, warnings }; }; diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index aeba61d6a..46c4605be 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -189,6 +189,36 @@ export function toNativeSafePath(p: string): string { return p; } +/** + * Resolve the on-disk CSV staging dir for `/`, applying the + * same ASCII-safe relocation `toNativeSafePath` enables: on Windows with a + * non-ASCII storage path, LadybugDB's bulk COPY cannot open files under that + * path, so the dir is relocated under `os.tmpdir()`. Shared by the structural + * `csv/` dir and the streaming `pdg-csv/` dir (#2202) so the two can never + * diverge on platform handling; the `gitnexus--` prefix keeps their tmp + * locations distinct and recognizable. + * + * The relocated dir is created with `fs.mkdtempSync` (a unique, mode-0700, + * guaranteed-not-pre-existing suffix) rather than a deterministic + * `gitnexus--` name. A predictable name in the world-readable OS + * temp dir is information-disclosure-prone and pre-plantable + * (CWE-377/378 / CodeQL `js/insecure-temporary-file`); mkdtemp's random suffix + * is the documented mitigation and is what reaches the streaming sink's + * `fs.openSync`. The non-Windows / ASCII path stays a pure `path.join` (no dir + * created) and is byte-identical to before. + */ +export function resolveNativeSafeStorageDir(storagePath: string, subdir: string): string { + if (process.platform === 'win32' && NON_ASCII_RE.test(storagePath)) { + // 8.3-shorten the tmpdir base first (a non-ASCII Windows *profile* path can + // make os.tmpdir() itself non-ASCII), THEN mkdtemp so the returned path — + // the one that flows into fs.openSync — is provably mkdtemp-sourced (random, + // exclusive) and clears the insecure-temp-file dataflow. + const base = toNativeSafePath(os.tmpdir()); + return fsSync.mkdtempSync(path.join(base, `gitnexus-${subdir}-`)); + } + return path.join(storagePath, subdir); +} + /** * Shared configuration for `@ladybugdb/core` `Database` construction. * diff --git a/gitnexus/src/core/lbug/pdg-emit-sink.ts b/gitnexus/src/core/lbug/pdg-emit-sink.ts new file mode 100644 index 000000000..79cc05f58 --- /dev/null +++ b/gitnexus/src/core/lbug/pdg-emit-sink.ts @@ -0,0 +1,395 @@ +/** + * Streaming PDG graph-emit sink (issue #2202). + * + * The PDG emit loop (`scope-resolution/pipeline/run.ts`, the `--pdg` block) + * materializes BasicBlock nodes + intra-file PDG edges (CFG / REACHING_DEF / + * CDG / POST_DOMINATE / TAINTED / SANITIZES) into the in-memory + * `KnowledgeGraph`. At full-kernel scale that layer dominates peak RSS + * (~7 GB at 511K BasicBlocks; ~100 GB extrapolated to the full kernel → OOM). + * + * `PdgEmitSink` is a write-routing façade over the real graph: the emit + * functions are write-only and compute every edge endpoint by deterministic id + * (audited — no read-back), so the sink can route BasicBlock node rows and PDG + * edge rows straight to bounded CSV-on-disk writers and **never store them**. + * The graph's resident size stops growing with the PDG layer → peak RSS becomes + * O(chunk buffer), not O(graph). Everything else (structural nodes/edges, the + * whole-program M4 TAINT_PATH edges) is delegated to the real graph unchanged. + * + * Why synchronous writers? The whole PDG emit (`runScopeResolution` and its + * per-file loop) is synchronous — there is no `await` point to drain an async + * stream, so a `BufferedCSVWriter` (Node `WriteStream`) would accumulate + * unwritten chunks in process memory across millions of rows, defeating the RSS + * bound. `fs.writeSync` goes straight to the OS; resident memory is bounded to + * one `chunkRows` buffer. This mirrors the sync-shard pattern in + * `storage/parsedfile-store.ts`. + * + * Byte-identity (issue acceptance): the sink reuses the SAME shared row + * builders (`buildBasicBlockRow`, `buildRelRow`) and label derivation + * (`getNodeLabel`) as `streamAllCSVsToDisk`, so the streamed CSV line SET is + * identical to the whole-graph emit's, and the bulk COPY loads the same rows → + * the persisted graph is SET-identical and DB-identical. The guarantee is + * set-level, not byte-level on the CSV file: the sink streams rows in emit + * order and does NOT re-sort them under `GITNEXUS_SORT_GRAPH_OUTPUT`, so a + * streamed CSV file is not necessarily byte-for-byte equal to the sorted + * whole-graph CSV — but the row set, and therefore the DB outcome, is. (The + * streamed CSVs are deleted right after the COPY, so their on-disk byte order + * is never observed.) Cross-pass dedup is done upstream, per FILE, in the emit + * loop (`run.ts` skips a file whose PDG already streamed) rather than in the + * sink, because a file can be PDG-emitted in more than one language pass (a + * `.ts` module imported by a `.vue` SFC is emitted in both the TypeScript and + * Vue context passes over the same worker-built CFG) and a sink-level per-id + * dedup set would retain every id → O(total ids) memory, defeating the + * O(chunk) RSS bound (#2202 review #1). The differential fingerprint test + * (issue #2202 U6) and the Vue+TS cross-pass integration test guard the set. + */ + +import fs from 'fs'; +import path from 'path'; +import type { GraphNode, GraphRelationship, RelationshipType } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../graph/types.js'; +import { + BASICBLOCK_CSV_HEADER, + REL_CSV_HEADER, + buildBasicBlockRow, + buildRelRow, +} from './csv-generator.js'; +import { getNodeLabel } from './rel-pair-routing.js'; +import { NODE_TABLES, type NodeTableName } from './schema.js'; + +/** + * PDG edge types streamed per-file (all intra-block BasicBlock→BasicBlock). + * `TAINT_PATH` is intentionally excluded — it is the whole-program M4 edge + * (Function→Function), computed in a separate post-resolution phase over the + * complete CALLS graph, and stays in the in-memory graph (it is small and is + * persisted by the normal whole-graph emit). + */ +const PDG_EDGE_TYPES: ReadonlySet = new Set([ + 'CFG', + 'REACHING_DEF', + 'CDG', + 'POST_DOMINATE', + 'TAINTED', + 'SANITIZES', +]); + +/** Default streamed-write buffer (rows). Matches the whole-graph emit's + * `FLUSH_EVERY` order of magnitude; overridable via `GITNEXUS_PDG_EMIT_CHUNK_SIZE`. */ +export const DEFAULT_PDG_EMIT_CHUNK_ROWS = 500; + +/** + * Synchronous buffered CSV writer. Buffers up to `chunkRows` rows, then issues + * one `fs.writeSync` straight to the OS (no in-process stream buffer). Header + * is written into the buffer at construction and is NOT counted in `rows` + * (matching `BufferedCSVWriter` semantics, so manifest row counts line up). + */ +class SyncCsvWriter { + private fd: number; + private buf: string[] = []; + private readonly chunkRows: number; + rows = 0; + /** + * First IO error this writer hit (a `fs.writeSync` short-write loop throwing + * on e.g. disk-full). Once poisoned the writer refuses further rows and + * skips its final flush; the sink surfaces it from {@link PdgEmitSink.finalize} + * so a truncated CSV is never handed to the bulk COPY (#2202 review #4). A + * streamed-write failure is an IO fault, not the CFG-logic error that the + * emit loop's per-file try/catch is built to swallow — poisoning routes it + * past that catch to a loud failure. + */ + poison: unknown | undefined = undefined; + + constructor( + readonly csvPath: string, + header: string, + chunkRows: number, + ) { + // Guard a 0/negative buffer: the flush modulo would never fire and `buf` + // would grow unbounded, defeating the whole point of streaming. + this.chunkRows = Math.max(1, chunkRows); + // Exclusive create (O_EXCL): the streamed-CSV dir is wiped + recreated fresh + // by the PdgEmitSink constructor before any writer opens a file, so the path + // never pre-exists — 'wx' both matches that invariant and refuses to follow + // a pre-planted symlink at the path (CWE-377 / CodeQL js/insecure-temporary-file). + this.fd = fs.openSync(csvPath, 'wx'); + this.buf.push(header); + } + + addRow(row: string): void { + // A poisoned writer is dead — stop buffering so memory can't grow on a + // writer whose fd is already in a bad state; finalize will report the fault. + if (this.poison !== undefined) return; + this.buf.push(row); + this.rows++; + // Flush on DATA-row count, not buffer length: the header occupies buf[0] + // until the first flush, so a `buf.length >= chunkRows` test would fire one + // row early on the first chunk. Counting rows makes every flush exactly + // `chunkRows` rows. + if (this.rows % this.chunkRows === 0) this.flushOrPoison(); + } + + /** Flush, recording (and re-throwing) any IO error as poison. Re-throwing + * lets the immediate caller log the per-file failure; the persisted `poison` + * is the backstop that makes finalize fail loudly even when that throw is + * swallowed by the emit loop's CFG try/catch. */ + private flushOrPoison(): void { + try { + this.flush(); + } catch (e) { + this.poison ??= e; + throw e; + } + } + + private flush(): void { + if (this.buf.length === 0) return; + const data = Buffer.from(this.buf.join('\n') + '\n', 'utf8'); + // fs.writeSync can return a short byte count; loop until the whole buffer + // lands so a partial write never truncates a CSV row mid-field. + let offset = 0; + while (offset < data.length) { + offset += fs.writeSync(this.fd, data, offset, data.length - offset); + } + this.buf.length = 0; + } + + /** Flush remaining rows (unless already poisoned) and close the fd. Never + * throws: a final-flush IO error is recorded as poison and the fd is still + * closed, so a write error neither leaks an fd nor escapes here — the sink + * reads {@link poison} after closing every writer and fails loudly then. */ + close(): void { + try { + if (this.poison === undefined) this.flush(); + } catch (e) { + this.poison ??= e; + } finally { + try { + fs.closeSync(this.fd); + } catch { + /* fd may already be invalid after an IO fault — nothing to recover */ + } + } + } +} + +/** + * COPY manifest produced by {@link PdgEmitSink.finalize}. Shaped to merge + * directly into `StreamedCSVResult` so `loadGraphToLbug` COPYs the streamed + * PDG CSVs through the same per-table / per-pair loops as the structural CSVs. + * Paths are absolute, so persistence needs no dir recomputation. + */ +export interface PdgEmitManifest { + /** Node-table CSVs (only `BasicBlock` today). */ + readonly nodeFiles: Map; + /** pairKey (`From|To`) → per-pair edge CSV. */ + readonly relsByPair: Map; +} + +/** + * Write-routing graph façade. Construct one per analyze run, thread it into the + * per-language `runScopeResolution` calls in place of the real graph during the + * `--pdg` emit, then {@link finalize} once after the last language. + */ +export class PdgEmitSink implements KnowledgeGraph { + private readonly validTables: Set; + private bbWriter: SyncCsvWriter | undefined; + /** pairKey (`From|To`) → writer. PDG edges are all `BasicBlock|BasicBlock`, + * but the map keeps the sink general and the manifest pair-keyed. */ + private readonly relWriters = new Map(); + private finalized = false; + /** + * First writer-construction failure (a `fs.openSync` throwing on e.g. EMFILE + * — out of file descriptors). The failure happens inside the `SyncCsvWriter` + * constructor before a writer object exists to carry poison, so it is held + * here at the sink level and folded into the {@link finalize} error check. + * Like an in-flight write fault, an open failure mid-emit would otherwise be + * swallowed by the emit loop's per-file try/catch and silently drop the rest + * of that file's rows (#2202 review #4/#6). + */ + private openFailure: unknown | undefined = undefined; + // NOTE on dedup: the same file can be PDG-emitted in more than one language + // pass (e.g. a `.ts` module imported by a `.vue` SFC is emitted in both the + // TypeScript pass and the Vue context pass over the same worker-built + // `cfgSideChannel`). The in-memory graph dedups that by id (first-writer-wins); + // this sink does NOT — to keep peak memory O(write buffer) rather than + // O(total ids), cross-pass dedup is done upstream, per FILE, in the emit loop + // (`run.ts` skips a file whose PDG already streamed via `pdgEmittedFiles`). + // The sink therefore receives each id exactly once and is a faithful + // pass-through; it must not be fed duplicate ids. + + constructor( + private readonly real: KnowledgeGraph, + private readonly pdgCsvDir: string, + private readonly chunkRows: number = DEFAULT_PDG_EMIT_CHUNK_ROWS, + ) { + this.validTables = new Set(NODE_TABLES as readonly string[]); + // Clear any streamed CSVs left by a previous (possibly crashed) run so a + // later COPY never picks up stale rows. + fs.rmSync(pdgCsvDir, { recursive: true, force: true }); + fs.mkdirSync(pdgCsvDir, { recursive: true }); + } + + // ── routed writes ────────────────────────────────────────────────────────── + + addNode(node: GraphNode): void { + if (node.label === 'BasicBlock') { + if (this.bbWriter === undefined) { + try { + this.bbWriter = new SyncCsvWriter( + path.join(this.pdgCsvDir, 'basicblock.csv'), + BASICBLOCK_CSV_HEADER, + this.chunkRows, + ); + } catch (e) { + this.openFailure ??= e; + throw e; + } + } + this.bbWriter.addRow(buildBasicBlockRow(node)); + return; + } + this.real.addNode(node); + } + + addRelationship(relationship: GraphRelationship): void { + if (PDG_EDGE_TYPES.has(relationship.type)) { + const fromLabel = getNodeLabel(relationship.sourceId); + const toLabel = getNodeLabel(relationship.targetId); + // Skip edges whose endpoint labels are not valid node tables — mirrors + // `RelPairRouter` exactly so the streamed set matches the whole-graph set. + if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) return; + const pairKey = `${fromLabel}|${toLabel}`; + let writer = this.relWriters.get(pairKey); + if (writer === undefined) { + try { + writer = new SyncCsvWriter( + path.join(this.pdgCsvDir, `rel_${fromLabel}_${toLabel}.csv`), + REL_CSV_HEADER, + this.chunkRows, + ); + } catch (e) { + this.openFailure ??= e; + throw e; + } + this.relWriters.set(pairKey, writer); + } + writer.addRow(buildRelRow(relationship)); + return; + } + this.real.addRelationship(relationship); + } + + /** Flush + close every streamed writer and return the COPY manifest. Every + * fd is closed even when a writer is poisoned (its `close` never throws); any + * IO fault — an in-flight write that poisoned a writer, a final-flush failure, + * or a writer-open failure (EMFILE) — is surfaced loudly here so a disk-full + * / out-of-fds run never hands a truncated CSV to the bulk COPY (#2202 review + * #4). The emit loop's per-file try/catch swallows the synchronous throw, so + * this poison check is the backstop that turns a silent partial manifest into + * a hard failure. */ + finalize(): PdgEmitManifest { + if (this.finalized) throw new Error('PdgEmitSink.finalize() called twice'); + this.finalized = true; + + const errors: unknown[] = []; + if (this.openFailure !== undefined) errors.push(this.openFailure); + + const nodeFiles = new Map(); + if (this.bbWriter !== undefined) { + this.bbWriter.close(); + if (this.bbWriter.poison !== undefined) errors.push(this.bbWriter.poison); + nodeFiles.set('BasicBlock' as NodeTableName, { + csvPath: this.bbWriter.csvPath, + rows: this.bbWriter.rows, + }); + } + + const relsByPair = new Map(); + for (const [pairKey, writer] of this.relWriters) { + writer.close(); + if (writer.poison !== undefined) errors.push(writer.poison); + relsByPair.set(pairKey, { csvPath: writer.csvPath, rows: writer.rows }); + } + + if (errors.length > 0) { + const first = errors[0]; + throw new Error( + `PdgEmitSink: ${errors.length} streamed CSV writer(s) hit an IO error ` + + `(disk-full / out-of-fds) during the emit — the persisted graph would ` + + `be truncated, so the run is failed rather than COPYing a partial CSV: ${ + first instanceof Error ? first.message : String(first) + }`, + ); + } + + return { nodeFiles, relsByPair }; + } + + /** + * Best-effort fd release for the error path — when a language pass throws + * before {@link finalize} runs, the caller's `finally` calls this so the + * BasicBlock + per-pair fds never leak. Idempotent with finalize via the + * `finalized` flag; close errors are swallowed because the run is already + * failing. + */ + close(): void { + if (this.finalized) return; + this.finalized = true; + try { + this.bbWriter?.close(); + } catch { + /* best-effort */ + } + for (const writer of this.relWriters.values()) { + try { + writer.close(); + } catch { + /* best-effort */ + } + } + } + + // ── delegated reads / non-PDG mutations ───────────────────────────────────── + // The PDG emit functions never call these on the routed graph, but the + // façade implements the full KnowledgeGraph surface so it is a drop-in for + // the emit target and any non-PDG write transparently reaches the real graph. + + get nodes(): GraphNode[] { + return this.real.nodes; + } + get relationships(): GraphRelationship[] { + return this.real.relationships; + } + iterNodes(): IterableIterator { + return this.real.iterNodes(); + } + iterRelationships(): IterableIterator { + return this.real.iterRelationships(); + } + iterRelationshipsByType(type: RelationshipType): IterableIterator { + return this.real.iterRelationshipsByType(type); + } + forEachNode(fn: (node: GraphNode) => void): void { + this.real.forEachNode(fn); + } + forEachRelationship(fn: (rel: GraphRelationship) => void): void { + this.real.forEachRelationship(fn); + } + getNode(id: string): GraphNode | undefined { + return this.real.getNode(id); + } + get nodeCount(): number { + return this.real.nodeCount; + } + get relationshipCount(): number { + return this.real.relationshipCount; + } + removeNode(nodeId: string): boolean { + return this.real.removeNode(nodeId); + } + removeNodesByFile(filePath: string): number { + return this.real.removeNodesByFile(filePath); + } + removeRelationship(relationshipId: string): boolean { + return this.real.removeRelationship(relationshipId); + } +} diff --git a/gitnexus/src/core/lbug/rel-pair-routing.ts b/gitnexus/src/core/lbug/rel-pair-routing.ts new file mode 100644 index 000000000..a10a817fa --- /dev/null +++ b/gitnexus/src/core/lbug/rel-pair-routing.ts @@ -0,0 +1,159 @@ +/** + * Relationship per-label-pair routing (#2203 U2). + * + * LadybugDB's bulk `COPY` into the single `CodeRelation` rel table requires a + * separate CSV per FROM→TO node-label pair (the `from=`/`to=` COPY params). + * Historically the emit pass wrote one monolithic `relations.csv`, which + * `loadGraphToLbug` then RE-READ line-by-line (regex per edge) and re-split + * into per-pair files — writing and reading the entire ~1M-edge set twice. + * + * This router lets the single emit pass route each edge to its per-pair file + * directly, so the monolithic write + re-read + per-edge regex are all gone. + * The label-derivation + validTables filtering + per-pair-file format here match + * the legacy `splitRelCsvByLabelPair`, so the per-pair files are byte-identical + * for all quote-free ids — see the differential test in + * `test/integration/csv-pipeline.test.ts`. ONE intentional divergence: this + * router derives the label from the RAW id, while the oracle re-derives it via a + * regex over the ESCAPED row — so for an id containing a `"` the router is the + * more-correct path (it routes the edge to the right pair; the oracle's regex + * mis-buckets or drops it). `splitRelCsvByLabelPair` is retained as the + * differential oracle (the quote-in-id divergence is asserted explicitly). + * + * Backpressure: at most one stream is awaited at a time (the caller routes + * edges sequentially and awaits the returned drain promise before the next), + * mirroring the legacy split's `for await` invariant. The hot path (existing + * pair, no backpressure) returns `void` — no microtask per edge. + */ +import path from 'path'; +import { createWriteStream, type WriteStream } from 'fs'; +import { once } from 'events'; +import { finished } from 'stream/promises'; + +/** Injectable for tests (backpressure/error simulation), mirroring split. */ +export type WriteStreamFactory = (filePath: string) => WriteStream; + +/** + * Derive a node's table label from its graph id. Matches the legacy + * `getNodeLabel` that lived inline in `loadGraphToLbug`: + * - `comm_*` → Community + * - `proc_*` → Process + * - otherwise the prefix before the first `:` (e.g. `Function:…` → Function) + */ +export const getNodeLabel = (nodeId: string): string => { + if (nodeId.startsWith('comm_')) return 'Community'; + if (nodeId.startsWith('proc_')) return 'Process'; + return nodeId.split(':')[0]; +}; + +export interface RelPairMeta { + csvPath: string; + rows: number; +} + +/** + * Routes already-escaped relationship CSV rows to per-FROM→TO-label-pair + * files. Filters edges whose endpoint labels are not valid node tables + * (counted as `skipped`), exactly as the legacy split did. + */ +export class RelPairRouter { + /** pairKey (`From|To`) → { csvPath, rows } */ + readonly byPair = new Map(); + private readonly streams = new Map(); + skipped = 0; + total = 0; + + private streamError: Error | null = null; + private readonly abort = new AbortController(); + + constructor( + private readonly csvDir: string, + private readonly header: string, + private readonly validTables: Set, + private readonly wsFactory: WriteStreamFactory = (p) => createWriteStream(p, 'utf-8'), + ) {} + + private markError = (err: Error): void => { + this.streamError ??= err; + this.abort.abort(err); + }; + + /** + * The first stream error observed, if any. Lets the emit caller rethrow the + * real error (EMFILE / disk-full) instead of the generic `AbortError` that a + * pending `once(ws,'drain',{signal})` rejects with when the abort fires — + * mirroring the retained `splitRelCsvByLabelPair`'s `throw streamError ?? err`. + */ + get lastError(): Error | null { + return this.streamError; + } + + /** + * Route one already-escaped CSV row (no trailing newline) to its pair file. + * Returns `void` on the synchronous hot path; a `Promise` only when a + * stream signals backpressure (or a new pair's header does) — the caller + * awaits the promise before routing the next edge. + */ + route(fromId: string, toId: string, row: string): void | Promise { + if (this.streamError) throw this.streamError; + + const fromLabel = getNodeLabel(fromId); + const toLabel = getNodeLabel(toId); + if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) { + this.skipped++; + return; + } + + const pairKey = `${fromLabel}|${toLabel}`; + const ws = this.streams.get(pairKey); + if (ws === undefined) { + // First edge for this pair: open the stream, write header + row. + return this.openAndWrite(pairKey, fromLabel, toLabel, row); + } + + this.byPair.get(pairKey)!.rows++; + this.total++; + if (!ws.write(row + '\n')) { + return once(ws, 'drain', { signal: this.abort.signal }).then(() => undefined); + } + } + + private async openAndWrite( + pairKey: string, + fromLabel: string, + toLabel: string, + row: string, + ): Promise { + const csvPath = path.join(this.csvDir, `rel_${fromLabel}_${toLabel}.csv`); + const ws = this.wsFactory(csvPath); + ws.on('error', this.markError); + this.streams.set(pairKey, ws); + this.byPair.set(pairKey, { csvPath, rows: 1 }); + this.total++; + if (!ws.write(this.header + '\n')) { + await once(ws, 'drain', { signal: this.abort.signal }); + } + if (!ws.write(row + '\n')) { + await once(ws, 'drain', { signal: this.abort.signal }); + } + } + + /** Flush + close every pair stream. Rejects if any stream errored. */ + async close(): Promise { + if (this.streamError) { + this.destroy(); + throw this.streamError; + } + await Promise.all( + Array.from(this.streams.values()).map(async (ws) => { + ws.end(); + await finished(ws); + }), + ); + if (this.streamError) throw this.streamError; + } + + /** Tear down all streams (no flush) — used on the error path. */ + destroy(): void { + for (const ws of this.streams.values()) ws.destroy(); + } +} diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 769125ad5..ed92c9b9c 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -61,6 +61,7 @@ import { } from './ingestion/taint/interproc-solver.js'; import { DEFAULT_PDG_MAX_INTERPROC_EDGES } from './ingestion/taint/interproc-emit.js'; import { taintModelVersion } from './ingestion/taint/typescript-model.js'; +import { parseTruthyEnv, parsePositiveIntEnv } from './ingestion/utils/env.js'; import { computeFileHashes, diffFileHashes } from '../storage/file-hash.js'; import { extractChangedSubgraph, @@ -170,6 +171,19 @@ export interface AnalyzeOptions { pdgMaxInterprocFindings?: number; pdgMaxInterprocHops?: number; pdgMaxInterprocEdges?: number; + /** + * Stream the BasicBlock + intra-file PDG-edge layer to CSV-on-disk during the + * emit loop instead of materializing it in the in-memory graph, bounding peak + * RSS to O(chunk) for full-kernel-scale repos (#2202). Only engages on a full + * rebuild — `resolveStreamPdgEmit` additionally requires `force === true` + * (the pre-pipeline guarantee of a full rebuild). May also be enabled via + * `GITNEXUS_STREAM_PDG_EMIT`. Memory-only; byte-identical output; not stamped + * into `RepoMeta.pdg`. */ + streamPdgEmit?: boolean; + /** Streamed PDG-emit write buffer (rows). `undefined` ⇒ + * `DEFAULT_PDG_EMIT_CHUNK_ROWS`. May also be set via + * `GITNEXUS_PDG_EMIT_CHUNK_SIZE`. Memory-only (#2202). */ + pdgEmitChunkSize?: number; /** * Default branch threaded into generated AGENTS.md / CLAUDE.md so the * regression-compare example uses the configured branch instead of a @@ -426,6 +440,48 @@ export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] => } : undefined; +/** + * Whether streaming/chunked PDG graph emit (#2202) engages this run. + * + * Streaming flushes the BasicBlock + intra-file PDG-edge layer to CSV-on-disk + * during the emit loop and never lands it in the in-memory graph, bounding peak + * RSS to O(chunk). It is sound ONLY on a full rebuild: the incremental + * writeback (`extractChangedSubgraph`) reads BasicBlock nodes back out of the + * in-memory graph, which streaming has already offloaded. `force === true` is + * the pre-pipeline guarantee of a full rebuild — `isIncremental` has + * `!force` as a necessary condition — so gating on it avoids the deliberately + * absent pre-pipeline incremental prediction (see the `isIncremental` note). + * + * Requires `pdg === true` (nothing to stream otherwise). Enabled by either the + * explicit `streamPdgEmit` option or the `GITNEXUS_STREAM_PDG_EMIT` env toggle. + * Memory-only — NOT part of {@link resolvePdgConfig}, so toggling it never + * trips `pdgModeMismatch`. Read every call (not memoized) so `vi.stubEnv` + * works in tests. Pure + exported for testing. + */ +export const resolveStreamPdgEmit = (options: { + pdg?: boolean; + force?: boolean; + streamPdgEmit?: boolean; +}): boolean => + options.pdg === true && + options.force === true && + (options.streamPdgEmit === true || parseTruthyEnv(process.env.GITNEXUS_STREAM_PDG_EMIT)); + +/** + * Resolve the streamed PDG-emit write-buffer size (#2202). Explicit option wins + * over `GITNEXUS_PDG_EMIT_CHUNK_SIZE`; `undefined` ⇒ the sink's + * `DEFAULT_PDG_EMIT_CHUNK_ROWS`. Memory-only; does not affect emitted bytes. + */ +export const resolvePdgEmitChunkSize = (options: { + pdgEmitChunkSize?: number; +}): number | undefined => { + // Only honor a positive-integer explicit option; `0`/negative is NOT nullish + // so `?? env` would pass it through and make the sink flush every row. + const opt = options.pdgEmitChunkSize; + if (opt !== undefined && Number.isInteger(opt) && opt > 0) return opt; + return parsePositiveIntEnv(process.env.GITNEXUS_PDG_EMIT_CHUNK_SIZE); +}; + /** * Whether the requested `--pdg` configuration differs from the one the * existing index's DB rows were built under (#2099 F1). An absent recorded @@ -828,6 +884,11 @@ export async function runFullAnalysis( pdgMaxInterprocFindings: options.pdgMaxInterprocFindings, pdgMaxInterprocHops: options.pdgMaxInterprocHops, pdgMaxInterprocEdges: options.pdgMaxInterprocEdges, + // Streaming/chunked PDG emit (#2202) — gated to full-rebuild runs + // (force === true) so the incremental writeback never reads back an + // offloaded BasicBlock layer. Memory-only; byte-identical output. + streamPdgEmit: resolveStreamPdgEmit(options), + pdgEmitChunkSize: resolvePdgEmitChunkSize(options), fetchWrappers: options.fetchWrappers, }, ); @@ -1069,11 +1130,21 @@ export async function runFullAnalysis( }); } else { // ── Full rebuild ─────────────────────────────────────────────── - await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => { - lbugMsgCount++; - const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24)); - progress('lbug', pct, msg); - }); + // Pass the streamed PDG-emit manifest (#2202) so the BasicBlock layer that + // was flushed to CSV during the emit loop is COPY'd alongside the + // structural CSVs. Only ever set on a full rebuild (streaming is + // force-gated), so the incremental branch above never carries it. + await loadGraphToLbug( + pipelineResult.graph, + pipelineResult.repoPath, + storagePath, + (msg) => { + lbugMsgCount++; + const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24)); + progress('lbug', pct, msg); + }, + pipelineResult.pdgEmitManifest, + ); } // ── Phase 3: FTS (85–90%) ───────────────────────────────────────── diff --git a/gitnexus/src/types/pipeline.ts b/gitnexus/src/types/pipeline.ts index 5ad08cae6..4cbb28886 100644 --- a/gitnexus/src/types/pipeline.ts +++ b/gitnexus/src/types/pipeline.ts @@ -2,6 +2,7 @@ import type { KnowledgeGraph } from '../core/graph/types.js'; import { CommunityDetectionResult } from '../core/ingestion/community-processor.js'; import { ProcessDetectionResult } from '../core/ingestion/process-processor.js'; import type { ResolutionOutcome } from '../core/ingestion/scope-resolution/resolution-outcome.js'; +import type { PdgEmitManifest } from '../core/lbug/pdg-emit-sink.js'; // CLI-specific: in-memory result with graph + detection results export interface PipelineResult { @@ -27,4 +28,12 @@ export interface PipelineResult { * affordance so regression suites can prove the pool engaged. */ usedWorkerPool: boolean; + /** + * Streamed PDG-emit COPY manifest (#2202). Present only when streaming/chunked + * PDG emit was active (full rebuild + `--pdg` + enabled): the BasicBlock node + * CSV + per-pair PDG-edge CSVs flushed to disk during the emit loop, for + * `loadGraphToLbug` to COPY alongside the structural CSVs. Absent ⇒ the PDG + * layer (if any) is resident in `graph` and persists via the whole-graph emit. + */ + pdgEmitManifest?: PdgEmitManifest; } diff --git a/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue b/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue new file mode 100644 index 000000000..b7c100e4a --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue @@ -0,0 +1,28 @@ + + + + diff --git a/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts b/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts new file mode 100644 index 000000000..6bb7197c9 --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts @@ -0,0 +1,44 @@ +// Shared TypeScript module imported by app.vue. Because app.vue's +// `collectScopeContextPaths` does a transitive import closure, THIS file is +// pulled into the Vue scope-resolution pass IN ADDITION to the primary +// TypeScript pass — so its worker-built CFG (the functions below) is +// PDG-emitted in BOTH passes over the same `cfgSideChannel`, producing +// identical BasicBlock + PDG-edge ids. The in-memory graph dedups those by id +// (first-writer-wins Map); the streaming sink relies on per-file dedup in +// run.ts (`pdgEmittedFiles`). This is the real cross-pass double-emit the +// #2202 streaming dedup must collapse (review #8a). + +export function classify(x: number): string { + let label: string; + if (x > 0) { + label = 'positive'; + } else if (x < 0) { + label = 'negative'; + } else { + label = 'zero'; + } + return label; +} + +export function accumulate(n: number): number { + let sum = 0; + for (let i = 0; i < n; i++) { + if (i % 2 === 0) { + sum += i; + } else { + sum -= 1; + } + } + return sum; +} + +export function guard(value: number): number { + if (value > 100) { + return clamp(value); + } + return value; +} + +function clamp(value: number): number { + return value > 100 ? 100 : value; +} diff --git a/gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts b/gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts new file mode 100644 index 000000000..ddce07071 --- /dev/null +++ b/gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts @@ -0,0 +1,167 @@ +/** + * End-to-end proof of streaming/chunked PDG graph emit (issue #2202). + * + * Runs the real pipeline (workers + scope-resolution) on the pdg-repo fixture + * TWICE — a non-streamed baseline and a streamed run — and asserts: + * - the streamed run's in-memory graph holds ZERO BasicBlock nodes and ZERO + * intra-file PDG edges (the bulky layer was flushed to CSV, never resident + * — the O(chunk) RSS bound, R1); + * - the streamed run produces a `pdgEmitManifest` whose BasicBlock + PDG-edge + * row counts EQUAL the baseline's resident counts (same emitted SET, R2); + * - the baseline (streaming off) still emits the PDG layer into the graph, + * i.e. the default path is unchanged (R3). + * + * Both runs use a durable parse cache (streaming requires `parseCache.storagePath` + * for its CSV dir), differing ONLY in `streamPdgEmit`, so streaming is the only + * variable. + */ +import { describe, it, expect, afterAll } from 'vitest'; +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { runPipelineFromRepo } from '../../../src/core/ingestion/pipeline.js'; +import { loadParseCache } from '../../../src/storage/parse-cache.js'; +import type { PipelineResult } from '../../../src/types/pipeline.js'; + +const FIXTURE = path.join(__dirname, 'fixtures', 'pdg-repo'); +// A `.vue` SFC importing a `.ts` module: the TS module is PDG-emitted in BOTH +// the TypeScript pass and the Vue context pass (review #8a, cross-pass dedup). +const VUE_TS_FIXTURE = path.join(__dirname, 'fixtures', 'vue-ts-pdg'); +const PDG_EDGE_TYPES = new Set([ + 'CFG', + 'REACHING_DEF', + 'CDG', + 'POST_DOMINATE', + 'TAINTED', + 'SANITIZES', +]); + +const tmpDirs: string[] = []; +function freshRepo(fixture: string = FIXTURE): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-pdg-stream-')); + fs.cpSync(fixture, dir, { recursive: true }); + tmpDirs.push(dir); + return dir; +} +function freshStorage(): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-pdg-store-')); + tmpDirs.push(dir); + return dir; +} + +function pdgCounts(result: PipelineResult): { basicBlocks: number; pdgEdges: number } { + let basicBlocks = 0; + result.graph.forEachNode((n) => { + if (n.label === 'BasicBlock') basicBlocks++; + }); + let pdgEdges = 0; + for (const rel of result.graph.iterRelationships()) { + if (PDG_EDGE_TYPES.has(rel.type)) pdgEdges++; + } + return { basicBlocks, pdgEdges }; +} + +describe('#2202 — streaming PDG emit end-to-end', () => { + afterAll(() => { + for (const d of tmpDirs) fs.rmSync(d, { recursive: true, force: true }); + }); + + it('streams the PDG layer out of the graph while preserving the emitted set', async () => { + // ── Baseline: --pdg on, streaming OFF (durable cache, same as streamed) ── + const baseStorage = freshStorage(); + const baseline = await runPipelineFromRepo(freshRepo(), () => {}, { + pdg: true, + parseCache: await loadParseCache(baseStorage), + }); + const base = pdgCounts(baseline); + // R3 / sanity: the default path still materializes the PDG layer in-graph. + expect(base.basicBlocks).toBeGreaterThan(0); + expect(base.pdgEdges).toBeGreaterThan(0); + expect(baseline.pdgEmitManifest).toBeUndefined(); + + // ── Streamed: --pdg on, streaming ON ───────────────────────────────── + const streamStorage = freshStorage(); + const streamed = await runPipelineFromRepo(freshRepo(), () => {}, { + pdg: true, + streamPdgEmit: true, + parseCache: await loadParseCache(streamStorage), + }); + const streamedCounts = pdgCounts(streamed); + + // R1: the bulky PDG layer never accumulated in the in-memory graph. + expect(streamedCounts.basicBlocks).toBe(0); + expect(streamedCounts.pdgEdges).toBe(0); + + // R2: the streamed manifest carries the SAME emitted set as the baseline. + const manifest = streamed.pdgEmitManifest; + expect(manifest).toBeDefined(); + const bbRows = manifest!.nodeFiles.get('BasicBlock')?.rows ?? 0; + expect(bbRows).toBe(base.basicBlocks); + let manifestEdgeRows = 0; + for (const [, meta] of manifest!.relsByPair) manifestEdgeRows += meta.rows; + expect(manifestEdgeRows).toBe(base.pdgEdges); + + // The streamed BasicBlock CSV exists on disk under the storage dir. + const bbCsv = manifest!.nodeFiles.get('BasicBlock')?.csvPath; + expect(bbCsv).toBeDefined(); + expect(fs.existsSync(bbCsv!)).toBe(true); + }); + + it('collapses the real Vue+TS cross-pass double-emit to one streamed copy (review #8a)', async () => { + // A `.ts` module imported by a `.vue` SFC is PDG-emitted in BOTH the + // TypeScript pass and the Vue context pass (the Vue provider's + // `collectScopeContextPaths` follows the import) over the same worker-built + // `cfgSideChannel` → identical ids. The in-memory graph dedups those by id + // (Map first-writer-wins); the streaming sink is dedup-free, so the emit + // loop dedups per FILE via `pdgEmittedFiles`. Without that dedup the streamed + // manifest would carry shared.ts's blocks TWICE (verified out-of-band: 33 → + // 61 BasicBlock rows). This asserts the streamed SET equals the Map-deduped + // baseline — the load-bearing dedup regression guard. + const baseStorage = freshStorage(); + const baseline = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, { + pdg: true, + parseCache: await loadParseCache(baseStorage), + }); + const base = pdgCounts(baseline); + // Both files contribute blocks — the cross-pass case is actually present. + expect(base.basicBlocks).toBeGreaterThan(0); + expect(base.pdgEdges).toBeGreaterThan(0); + + const streamStorage = freshStorage(); + const streamed = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, { + pdg: true, + streamPdgEmit: true, + parseCache: await loadParseCache(streamStorage), + }); + // Streamed graph holds none of the PDG layer. + const streamedCounts = pdgCounts(streamed); + expect(streamedCounts.basicBlocks).toBe(0); + expect(streamedCounts.pdgEdges).toBe(0); + + // The streamed manifest carries each file's PDG layer EXACTLY ONCE — equal + // to the Map-deduped baseline. A broken per-file dedup would double the + // shared module's rows and fail here. + const manifest = streamed.pdgEmitManifest; + expect(manifest).toBeDefined(); + expect(manifest!.nodeFiles.get('BasicBlock')?.rows ?? 0).toBe(base.basicBlocks); + let manifestEdgeRows = 0; + for (const [, meta] of manifest!.relsByPair) manifestEdgeRows += meta.rows; + expect(manifestEdgeRows).toBe(base.pdgEdges); + }); + + it('falls back to in-memory emit when streaming is on but no storage path exists', async () => { + // `streamPdgEmit: true` with NO parse cache → `parsedFileStorePath` is + // undefined, so phase.ts cannot place the streamed CSV dir and falls back to + // the in-memory whole-graph emit (the `else` branch that warns). The PDG + // layer must still land in the graph and NO manifest is produced. + const fellBack = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, { + pdg: true, + streamPdgEmit: true, + // intentionally no parseCache → no storagePath + }); + const counts = pdgCounts(fellBack); + expect(counts.basicBlocks).toBeGreaterThan(0); // emitted in-memory, not streamed + expect(counts.pdgEdges).toBeGreaterThan(0); + expect(fellBack.pdgEmitManifest).toBeUndefined(); // no streaming happened + }); +}); diff --git a/gitnexus/test/integration/csv-pipeline.test.ts b/gitnexus/test/integration/csv-pipeline.test.ts index cf44c85b2..1d3ee0e7c 100644 --- a/gitnexus/test/integration/csv-pipeline.test.ts +++ b/gitnexus/test/integration/csv-pipeline.test.ts @@ -4,17 +4,45 @@ * Tests: streamAllCSVsToDisk with real graph data. * Covers hardening fixes: LRU cache (#24), BufferedCSVWriter flush */ -import { describe, it, expect, beforeAll, afterAll } from 'vitest'; +import { describe, it, expect, beforeAll, beforeEach, afterAll } from 'vitest'; import fs from 'fs/promises'; +import { finished } from 'stream/promises'; import path from 'path'; import { createTempDir, type TestDBHandle } from '../helpers/test-db.js'; import { buildTestGraph, type TestNodeInput, type TestRelInput } from '../helpers/test-graph.js'; -import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.js'; +import { + streamAllCSVsToDisk, + buildRelRow, + REL_CSV_HEADER, +} from '../../src/core/lbug/csv-generator.js'; +import { splitRelCsvByLabelPair } from '../../src/core/lbug/lbug-adapter.js'; +import { getNodeLabel } from '../../src/core/lbug/rel-pair-routing.js'; +import { NODE_TABLES } from '../../src/core/lbug/schema.js'; let tmpHandle: TestDBHandle; let csvDir: string; let repoDir: string; +/** Data rows (header dropped) of one CSV file's text. */ +const dataRowsOf = (csv: string): string[] => + csv + .trim() + .split('\n') + .slice(1) + .filter((l) => l.length > 0); + +/** Concatenate data rows from every per-pair rel file (#2203 U2), pair keys + * sorted so the concatenation order is deterministic regardless of map order. */ +const readAllRelRows = async ( + relsByPair: Map, +): Promise => { + const rows: string[] = []; + for (const key of [...relsByPair.keys()].sort()) { + rows.push(...dataRowsOf(await fs.readFile(relsByPair.get(key)!.csvPath, 'utf-8'))); + } + return rows; +}; + beforeAll(async () => { tmpHandle = await createTempDir('csv-pipeline-test-'); csvDir = path.join(tmpHandle.dbPath, 'csv'); @@ -76,9 +104,9 @@ describe('streamAllCSVsToDisk', () => { { id: 'folder:src', label: 'Folder', name: 'src', filePath: 'src' }, ], [ - { sourceId: 'func:main', targetId: 'func:helper', type: 'CALLS' }, - { sourceId: 'file:src/index.ts', targetId: 'func:main', type: 'CONTAINS' }, - { sourceId: 'file:src/utils.ts', targetId: 'func:helper', type: 'CONTAINS' }, + { sourceId: 'Function:main', targetId: 'Function:helper', type: 'CALLS' }, + { sourceId: 'File:src/index.ts', targetId: 'Function:main', type: 'CONTAINS' }, + { sourceId: 'File:src/utils.ts', targetId: 'Function:helper', type: 'CONTAINS' }, ], ); @@ -86,7 +114,8 @@ describe('streamAllCSVsToDisk', () => { // Check that CSV files were created expect(result.nodeFiles.size).toBeGreaterThan(0); - expect(result.relRows).toBe(3); + expect(result.totalValidRels).toBe(3); + expect(result.skippedRels).toBe(0); // Verify File CSV const fileCsv = result.nodeFiles.get('File'); @@ -108,10 +137,12 @@ describe('streamAllCSVsToDisk', () => { expect(folderCsv).toBeDefined(); expect(folderCsv!.rows).toBe(1); - // Verify relations CSV exists - const relContent = await fs.readFile(result.relCsvPath, 'utf-8'); - const relLines = relContent.trim().split('\n'); - expect(relLines.length).toBe(4); // header + 3 relationships + // Relationships are routed to per-FROM→TO-label-pair files (#2203 U2): + // Function→Function (CALLS) + File→Function (2× CONTAINS). + expect(result.relsByPair.has('Function|Function')).toBe(true); + expect(result.relsByPair.has('File|Function')).toBe(true); + expect(result.relsByPair.get('File|Function')!.rows).toBe(2); + expect(await readAllRelRows(result.relsByPair)).toHaveLength(3); }); it('CSV content is properly escaped', async () => { @@ -210,7 +241,8 @@ describe('streamAllCSVsToDisk', () => { const graph = buildTestGraph([], []); const result = await streamAllCSVsToDisk(graph, repoDir, csvDir); expect(result.nodeFiles.size).toBe(0); - expect(result.relRows).toBe(0); + expect(result.totalValidRels).toBe(0); + expect(result.relsByPair.size).toBe(0); }); it('handles node with empty string properties', async () => { @@ -221,6 +253,27 @@ describe('streamAllCSVsToDisk', () => { expect(fileCsv).toBeDefined(); expect(fileCsv!.rows).toBe(1); }); + + it('crosses the BufferedCSVWriter FLUSH_EVERY boundary without losing rows', async () => { + // FLUSH_EVERY=500; a >500-node graph forces ≥1 mid-stream flush, exercising + // addRow's flush-promise return + the loop's `if (pending) await pending` + // path that the small fixtures above never reach (only the bench did). + const N = 600; + const nodes = Array.from({ length: N }, (_, i) => ({ + id: `File:src/f${i}.ts`, + label: 'File' as const, + name: `f${i}.ts`, + filePath: `src/f${i}.ts`, + })); + const result = await streamAllCSVsToDisk(buildTestGraph(nodes), repoDir, csvDir); + + const fileCsv = result.nodeFiles.get('File'); + expect(fileCsv).toBeDefined(); + expect(fileCsv!.rows).toBe(N); // no rows dropped/duplicated at the flush boundary + const dataRows = dataRowsOf(await fs.readFile(fileCsv!.csvPath, 'utf-8')); + expect(dataRows).toHaveLength(N); + expect(new Set(dataRows).size).toBe(N); // all distinct — no flush-boundary corruption + }); }); /** @@ -234,15 +287,18 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => { // Folder nodes: single-line CSV rows (no multi-line `content` column), so the // id is the first comma-separated field and split('\n') is safe. ids are // deliberately NOT in insertion order (c, a, b). + // ids use the `Folder:` prefix so getNodeLabel derives the valid `Folder` + // table — edges route to rel_Folder_Folder.csv (#2203 U2). Deliberately NOT + // in insertion order (c, a, b). const NODES: TestNodeInput[] = [ - { id: 'folder:c', label: 'Folder', name: 'c', filePath: 'c' }, - { id: 'folder:a', label: 'Folder', name: 'a', filePath: 'a' }, - { id: 'folder:b', label: 'Folder', name: 'b', filePath: 'b' }, + { id: 'Folder:c', label: 'Folder', name: 'c', filePath: 'c' }, + { id: 'Folder:a', label: 'Folder', name: 'a', filePath: 'a' }, + { id: 'Folder:b', label: 'Folder', name: 'b', filePath: 'b' }, ]; const RELS: TestRelInput[] = [ - { sourceId: 'folder:c', targetId: 'folder:a', type: 'CONTAINS' }, - { sourceId: 'folder:a', targetId: 'folder:b', type: 'CONTAINS' }, - { sourceId: 'folder:b', targetId: 'folder:c', type: 'CONTAINS' }, + { sourceId: 'Folder:c', targetId: 'Folder:a', type: 'CONTAINS' }, + { sourceId: 'Folder:a', targetId: 'Folder:b', type: 'CONTAINS' }, + { sourceId: 'Folder:b', targetId: 'Folder:c', type: 'CONTAINS' }, ]; const dataRows = (csv: string): string[] => csv @@ -270,7 +326,7 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => { const folderIds = folderCsv ? dataRows(await fs.readFile(folderCsv.csvPath, 'utf-8')).map(firstCol) : []; - const relRows = dataRows(await fs.readFile(result.relCsvPath, 'utf-8')); + const relRows = await readAllRelRows(result.relsByPair); return { folderIds, relRows }; } finally { delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT; @@ -307,3 +363,205 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => { expect([...onFwd.relRows].sort()).toEqual([...offFwd.relRows].sort()); }); }); + +/** + * #2203 U2 byte-identity: for all quote-free ids the direct per-pair emit must + * produce per-pair files byte-for-byte identical to the legacy + * splitRelCsvByLabelPair oracle run over an equivalent monolithic relations.csv + * from the same graph. This is the load-bearing guard for "byte-identical graph + * content" (issue acceptance). The ONE intentional divergence — ids containing a + * double-quote, where the router (raw-id label) is more correct than the oracle + * (regex over the escaped row) — is asserted explicitly in its own test below. + */ +describe('streamAllCSVsToDisk — direct per-pair emit matches the split oracle', () => { + // The oracle always emits in graph.iterRelationships() (unsorted) order; the + // production path honours GITNEXUS_SORT_GRAPH_OUTPUT. Clear it so a value + // leaked from a prior test can't desync the two and produce a spurious diff. + beforeEach(() => { + delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT; + }); + + it('produces byte-identical per-pair files + identical skip/total accounting', async () => { + // Multiple valid pairs, getNodeLabel special prefixes (comm_ AND proc_), and + // one invalid-label edge that BOTH paths must skip identically. + const graph = buildTestGraph( + [ + { id: 'File:a.ts', label: 'File', name: 'a.ts', filePath: 'a.ts' }, + { id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' }, + { id: 'Function:a.ts:g:5', label: 'Function', name: 'g', filePath: 'a.ts' }, + { id: 'comm_1', label: 'Community' as never, name: 'c1', filePath: '' }, + { id: 'comm_2', label: 'Community' as never, name: 'c2', filePath: '' }, + { id: 'proc_1', label: 'Process' as never, name: 'p1', filePath: '' }, + { id: 'proc_2', label: 'Process' as never, name: 'p2', filePath: '' }, + ], + [ + { sourceId: 'File:a.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' }, + { sourceId: 'File:a.ts', targetId: 'Function:a.ts:g:5', type: 'CONTAINS' }, + { sourceId: 'Function:a.ts:f:1', targetId: 'Function:a.ts:g:5', type: 'CALLS' }, + { sourceId: 'comm_1', targetId: 'comm_2', type: 'CONTAINS' }, + // proc_ prefix → Process label (getNodeLabel special case). + { sourceId: 'proc_1', targetId: 'proc_2', type: 'CONTAINS' }, + // Invalid FROM label ('Bogus' ∉ NODE_TABLES) — skipped by both paths. + { sourceId: 'Bogus:x', targetId: 'File:a.ts', type: 'CONTAINS' }, + // Invalid TO label — exercises the OTHER branch of the skip condition. + { sourceId: 'File:a.ts', targetId: 'Bogus:y', type: 'CONTAINS' }, + ], + ); + + const directDir = path.join(csvDir, 'diff-direct'); + const oracleDir = path.join(csvDir, 'diff-oracle'); + await fs.mkdir(oracleDir, { recursive: true }); + + // Direct emit (production path). + const direct = await streamAllCSVsToDisk(graph, repoDir, directDir); + + // Oracle: build the monolithic relations.csv this graph would have produced + // (same insertion order, same row bytes via buildRelRow), then split it. + const relCsv = path.join(oracleDir, 'relations.csv'); + const lines = [REL_CSV_HEADER]; + for (const rel of graph.iterRelationships()) lines.push(buildRelRow(rel)); + await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8'); + + const split = await splitRelCsvByLabelPair( + relCsv, + oracleDir, + new Set(NODE_TABLES), + getNodeLabel, + ); + await Promise.all( + Array.from(split.pairWriteStreams.values()).map(async (ws) => { + ws.end(); + await finished(ws); + }), + ); + + // Identical accounting. + expect(direct.totalValidRels).toBe(split.totalValidRels); + expect(direct.totalValidRels).toBe(5); + expect(direct.skippedRels).toBe(split.skippedRels); + expect(direct.skippedRels).toBe(2); // invalid-FROM + invalid-TO, both skipped + expect(direct.relHeader).toBe(split.relHeader); + + // Identical pair set. + expect([...direct.relsByPair.keys()].sort()).toEqual([...split.relsByPairMeta.keys()].sort()); + + // Byte-identical per-pair file contents. + for (const key of direct.relsByPair.keys()) { + const directContent = await fs.readFile(direct.relsByPair.get(key)!.csvPath, 'utf-8'); + const oracleContent = await fs.readFile(split.relsByPairMeta.get(key)!.csvPath, 'utf-8'); + expect(directContent, `pair ${key}`).toBe(oracleContent); + } + }); + + it('quote-in-id edge: router routes it (raw-id label) while the oracle drops it — intended divergence', async () => { + // A node id with an embedded double-quote (legal in a POSIX filePath). The + // router derives the label from the RAW id (`File`), so it routes the edge; + // the oracle re-derives the label via /"([^"]*)","([^"]*)"/ over the ESCAPED + // row (`"File:a""b.ts",...`), mis-reads the field, and drops it. This locks + // the intended divergence so a future change can't silently revert the + // router to the buggy regex semantics. + const graph = buildTestGraph( + [ + { id: 'File:clean.ts', label: 'File', name: 'clean.ts', filePath: 'clean.ts' }, + { id: 'File:a"b.ts', label: 'File', name: 'a"b.ts', filePath: 'a"b.ts' }, + { id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' }, + ], + [ + { sourceId: 'File:clean.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' }, + { sourceId: 'File:a"b.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' }, + ], + ); + + const directDir = path.join(csvDir, 'qd-direct'); + const oracleDir = path.join(csvDir, 'qd-oracle'); + await fs.mkdir(oracleDir, { recursive: true }); + + const direct = await streamAllCSVsToDisk(graph, repoDir, directDir); + + const relCsv = path.join(oracleDir, 'relations.csv'); + const lines = [REL_CSV_HEADER]; + for (const rel of graph.iterRelationships()) lines.push(buildRelRow(rel)); + await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8'); + const split = await splitRelCsvByLabelPair( + relCsv, + oracleDir, + new Set(NODE_TABLES), + getNodeLabel, + ); + await Promise.all( + Array.from(split.pairWriteStreams.values()).map(async (ws) => { + ws.end(); + await finished(ws); + }), + ); + + // Router routes BOTH edges — the raw-id label `File` is valid for both. + expect(direct.totalValidRels).toBe(2); + expect(direct.skippedRels).toBe(0); + expect(direct.relsByPair.get('File|Function')!.rows).toBe(2); + + // Oracle DIVERGES: its regex mis-reads the quote-in-id row and drops that + // edge, so it routes strictly fewer edges. Asserted robustly — we do NOT + // pin the oracle's exact mis-derived label. + expect(split.totalValidRels).toBeLessThan(direct.totalValidRels); + expect(split.skippedRels).toBeGreaterThan(direct.skippedRels); + }); + + it('sorted path (GITNEXUS_SORT_GRAPH_OUTPUT=1): per-pair files byte-identical to the oracle', async () => { + // The earlier differential test covers the default (insertion-order) path. + // Here the sorted emit path must also match the oracle — fed the SAME + // id-sorted order orderedRelationships() uses (sort by rel.id). + process.env.GITNEXUS_SORT_GRAPH_OUTPUT = '1'; + try { + const graph = buildTestGraph( + [ + { id: 'File:a.ts', label: 'File', name: 'a.ts', filePath: 'a.ts' }, + { id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' }, + { id: 'Function:a.ts:g:5', label: 'Function', name: 'g', filePath: 'a.ts' }, + ], + // Deliberately NOT in id-sorted order so the sort actually reorders rows. + [ + { sourceId: 'Function:a.ts:f:1', targetId: 'Function:a.ts:g:5', type: 'CALLS' }, + { sourceId: 'File:a.ts', targetId: 'Function:a.ts:g:5', type: 'CONTAINS' }, + { sourceId: 'File:a.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' }, + ], + ); + + const directDir = path.join(csvDir, 'sorted-direct'); + const oracleDir = path.join(csvDir, 'sorted-oracle'); + await fs.mkdir(oracleDir, { recursive: true }); + + const direct = await streamAllCSVsToDisk(graph, repoDir, directDir); + + // Oracle fed the same id-sorted order the sorted emit produces. + const sortedRels = [...graph.iterRelationships()].sort((a, b) => + a.id < b.id ? -1 : a.id > b.id ? 1 : 0, + ); + const relCsv = path.join(oracleDir, 'relations.csv'); + const lines = [REL_CSV_HEADER]; + for (const rel of sortedRels) lines.push(buildRelRow(rel)); + await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8'); + const split = await splitRelCsvByLabelPair( + relCsv, + oracleDir, + new Set(NODE_TABLES), + getNodeLabel, + ); + await Promise.all( + Array.from(split.pairWriteStreams.values()).map(async (ws) => { + ws.end(); + await finished(ws); + }), + ); + + expect([...direct.relsByPair.keys()].sort()).toEqual([...split.relsByPairMeta.keys()].sort()); + for (const key of direct.relsByPair.keys()) { + const directContent = await fs.readFile(direct.relsByPair.get(key)!.csvPath, 'utf-8'); + const oracleContent = await fs.readFile(split.relsByPairMeta.get(key)!.csvPath, 'utf-8'); + expect(directContent, `pair ${key} (sorted)`).toBe(oracleContent); + } + } finally { + delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT; + } + }); +}); diff --git a/gitnexus/test/integration/lbug-load-prof.test.ts b/gitnexus/test/integration/lbug-load-prof.test.ts new file mode 100644 index 000000000..f83cd75c2 --- /dev/null +++ b/gitnexus/test/integration/lbug-load-prof.test.ts @@ -0,0 +1,141 @@ +/** + * Integration test: PROF_LBUG_LOAD persistence-path profiling (#2203 U1). + * + * loadGraphToLbug is un-timed in production today; the analyze "emit" number + * is the scope-resolution emit bucket, not this CSV→COPY persistence path. + * U1 adds a zero-cost-when-off per-stage breakdown gated by PROF_LBUG_LOAD=1, + * mirroring the PROF_SCOPE_RESOLUTION pattern. These tests assert the gate: + * - flag off → no `[lbug-load prof]` line is logged, behaviour unchanged + * - flag on → exactly one summary line with every stage key + node/rel counts + * + * Needs a real LadybugDB connection (initLbug), so it lives under integration. + * Logger assertions use `_captureLogger()` — the exported `logger` is a Proxy + * over a lazily-built pino instance and is not directly spy-able. + */ +import { describe, it, expect, beforeAll, beforeEach, afterAll, afterEach } from 'vitest'; +import fs from 'fs/promises'; +import path from 'path'; +import os from 'os'; +import { buildTestGraph } from '../helpers/test-graph.js'; +import { _captureLogger, type LoggerCapture } from '../../src/core/logger.js'; + +let tmpBase: string; +let storagePath: string; +let dbPath: string; +let cap: LoggerCapture; + +const PROF_LINE = '[lbug-load prof]'; + +const profLines = (): string[] => + cap + .records() + .map((r) => (typeof r.msg === 'string' ? r.msg : '')) + .filter((msg) => msg.includes(PROF_LINE)); + +beforeAll(async () => { + tmpBase = path.join(os.tmpdir(), `gitnexus-lbug-prof-${Date.now()}-${process.pid}`); + storagePath = path.join(tmpBase, '.gitnexus'); + dbPath = path.join(storagePath, 'lbug'); + await fs.mkdir(dbPath, { recursive: true }); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.initLbug(dbPath); +}); + +beforeEach(() => { + cap = _captureLogger(); +}); + +afterEach(() => { + cap.restore(); + delete process.env.PROF_LBUG_LOAD; +}); + +afterAll(async () => { + try { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.closeLbug(); + } catch { + /* may not have opened */ + } + try { + await fs.rm(tmpBase, { recursive: true, force: true }); + } catch { + /* best-effort */ + } +}); + +describe('PROF_LBUG_LOAD persistence-path profiling (#2203 U1)', () => { + it('does NOT log a prof summary when the flag is unset', async () => { + delete process.env.PROF_LBUG_LOAD; + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + const graph = buildTestGraph( + [ + { id: 'File:src/off.ts', label: 'File', name: 'off.ts', filePath: 'src/off.ts' }, + { + id: 'Function:src/off.ts:offFn:1', + label: 'Function', + name: 'offFn', + filePath: 'src/off.ts', + startLine: 1, + endLine: 2, + }, + ], + [{ sourceId: 'File:src/off.ts', targetId: 'Function:src/off.ts:offFn:1', type: 'DEFINES' }], + ); + + const result = await adapter.loadGraphToLbug(graph, tmpBase, storagePath); + + expect(result.success).toBe(true); + expect(profLines()).toHaveLength(0); + }); + + it('logs exactly one summary line with all stage keys + counts when the flag is set', async () => { + process.env.PROF_LBUG_LOAD = '1'; + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + // Distinct ids from the flag-off graph so the COPY does not hit a + // PK-dup IGNORE_ERRORS retry on the shared singleton connection. + const graph = buildTestGraph( + [ + { id: 'File:src/on.ts', label: 'File', name: 'on.ts', filePath: 'src/on.ts' }, + { + id: 'Function:src/on.ts:onFn:1', + label: 'Function', + name: 'onFn', + filePath: 'src/on.ts', + startLine: 1, + endLine: 2, + }, + { + id: 'Class:src/on.ts:OnClass:5', + label: 'Class', + name: 'OnClass', + filePath: 'src/on.ts', + startLine: 5, + endLine: 8, + }, + ], + [ + { sourceId: 'File:src/on.ts', targetId: 'Function:src/on.ts:onFn:1', type: 'DEFINES' }, + { sourceId: 'File:src/on.ts', targetId: 'Class:src/on.ts:OnClass:5', type: 'DEFINES' }, + ], + ); + + const result = await adapter.loadGraphToLbug(graph, tmpBase, storagePath); + expect(result.success).toBe(true); + + const lines = profLines(); + expect(lines).toHaveLength(1); + + const line = lines[0]; + // Relationships are routed to per-pair files during csv-emit (#2203 U2), + // so there is no separate rel-split stage. + for (const key of ['csv-emit=', 'copy-nodes=', 'copy-rels=', 'fallback=', 'total=']) { + expect(line).toContain(key); + } + // 3 node rows (File, Function, Class), 2 valid rels emitted. + expect(line).toContain('(3 nodes, 2 rels)'); + }); +}); diff --git a/gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts b/gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts new file mode 100644 index 000000000..d29d9f524 --- /dev/null +++ b/gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts @@ -0,0 +1,181 @@ +/** + * Integration test: streamed PDG-emit manifest round-trips the bulk-COPY load + * path (issue #2202 U5). + * + * Simulates the streaming case end-to-end at the persistence boundary: the + * BasicBlock + intra-file PDG-edge layer is flushed to CSV by a real + * `PdgEmitSink` (so the in-memory graph holds ZERO BasicBlocks, exactly as in a + * streamed run), and `loadGraphToLbug` is handed the resulting manifest. Asserts + * the BasicBlock nodes + every PDG edge type land in the DB via the manifest, + * alongside the structural graph — and that there is no double-COPY. + */ +import { describe, it, expect, beforeAll, afterAll } from 'vitest'; +import fs from 'fs/promises'; +import path from 'path'; +import os from 'os'; + +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import { PdgEmitSink } from '../../src/core/lbug/pdg-emit-sink.js'; +import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; + +let tmpBase: string; +let storagePath: string; + +const FILE_ID = 'File:src/a.ts'; +const BB = (i: number) => `BasicBlock:src/a.ts:1:0:${i}`; +const PDG_TYPES = ['CFG', 'REACHING_DEF', 'CDG', 'POST_DOMINATE', 'TAINTED', 'SANITIZES'] as const; + +beforeAll(async () => { + // mkdtemp (unpredictable, unique) — not a predictable os-temp path. + tmpBase = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-pdg-stream-rt-')); + storagePath = path.join(tmpBase, '.gitnexus'); + await fs.mkdir(path.join(storagePath, 'lbug'), { recursive: true }); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.initLbug(path.join(storagePath, 'lbug')); + + // Structural graph — NO BasicBlock nodes (they were "streamed out"). + const graph = createKnowledgeGraph(); + graph.addNode({ + id: FILE_ID, + label: 'File', + properties: { name: 'a.ts', filePath: 'src/a.ts' }, + }); + + // Real sink → real manifest: route 3 BasicBlocks + one edge of each PDG type. + const sink = new PdgEmitSink(graph, path.join(storagePath, 'pdg-csv')); + for (let i = 0; i < 3; i++) { + const node: GraphNode = { + id: BB(i), + label: 'BasicBlock', + properties: { + name: '', + filePath: 'src/a.ts', + startLine: i + 1, + endLine: i + 2, + text: `b${i}`, + }, + }; + sink.addNode(node); + } + for (const type of PDG_TYPES) { + const rel: GraphRelationship = { + id: `${type}:0->1`, + sourceId: BB(0), + targetId: BB(1), + type, + confidence: 1, + reason: type === 'REACHING_DEF' ? 'x' : type === 'CDG' ? 'T' : `${type}-edge`, + }; + sink.addRelationship(rel); + } + const manifest = sink.finalize(); + + // The sink offloaded the whole PDG layer — the graph has only the File node. + expect(graph.nodeCount).toBe(1); + + await adapter.loadGraphToLbug(graph, tmpBase, storagePath, undefined, manifest); +}); + +afterAll(async () => { + try { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.closeLbug(); + } catch { + /* may not have opened */ + } + if (tmpBase) { + for (let attempt = 0; attempt < 5; attempt++) { + try { + await fs.rm(tmpBase, { recursive: true, force: true }); + return; + } catch { + if (attempt < 4) await new Promise((r) => setTimeout(r, 200 * (attempt + 1))); + } + } + } +}); + +describe('streamed PDG manifest → bulk COPY (#2202 U5)', () => { + it('BasicBlock nodes from the manifest land in the DB with span + text', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const rows = await adapter.executeQuery( + 'MATCH (n:BasicBlock) RETURN n.id AS id, n.text AS text, n.startLine AS startLine ORDER BY n.id', + ); + expect(rows).toHaveLength(3); + expect(rows[0].id).toBe(BB(0)); + expect(rows[0].text).toBe('b0'); + expect(Number(rows[0].startLine)).toBe(1); + }); + + it('the structural graph (File node) loaded alongside the manifest', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const rows = await adapter.executeQuery( + `MATCH (f:File {id: '${FILE_ID}'}) RETURN count(f) AS c`, + ); + expect(Number(rows[0].c)).toBe(1); + }); + + it('every PDG edge type round-trips via the manifest (no double-COPY)', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + for (const type of PDG_TYPES) { + const rows = await adapter.executeQuery( + `MATCH (:BasicBlock)-[r:CodeRelation {type: '${type}'}]->(:BasicBlock) RETURN count(r) AS c`, + ); + // Exactly one — not two (double-COPY would double these). + expect(Number(rows[0].c), `${type} should round-trip exactly once`).toBe(1); + } + }); + + it('REACHING_DEF carries its variable in reason (manifest path)', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const rows = await adapter.executeQuery( + "MATCH (a:BasicBlock)-[r:CodeRelation {type: 'REACHING_DEF', reason: 'x'}]->(b:BasicBlock) RETURN a.id AS from, b.id AS to", + ); + expect(rows).toHaveLength(1); + expect(rows[0].from).toBe(BB(0)); + expect(rows[0].to).toBe(BB(1)); + }); +}); + +describe('streamed PDG manifest → disjoint-key merge guard (#2202 review #3)', () => { + // The merge in loadGraphToLbug assumes the streamed manifest and the + // structural csvResult are disjoint: when streaming is on the in-memory graph + // holds ZERO BasicBlocks, so streamAllCSVsToDisk emits no basicblock.csv and + // the manifest is the sole source. A future BasicBlock-leak-into-graph would + // make both sides carry a "BasicBlock" entry; silently overwriting one CSV + // with the other would drop its rows. The guard fails loudly instead. + + it('throws when the manifest collides with a structural node CSV', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + // A graph that DOES contain a BasicBlock → streamAllCSVsToDisk emits a + // structural basicblock.csv (the invariant-violation scenario). + const leakyGraph = createKnowledgeGraph(); + leakyGraph.addNode({ + id: BB(0), + label: 'BasicBlock', + properties: { name: '', filePath: 'src/a.ts', startLine: 1, endLine: 2, text: 'leak' }, + }); + + // A manifest that ALSO declares a BasicBlock node CSV → disjoint-key clash. + const collideBase = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-pdg-collide-')); + const collideStorage = path.join(collideBase, '.gitnexus'); + await fs.mkdir(collideStorage, { recursive: true }); + const sink = new PdgEmitSink(createKnowledgeGraph(), path.join(collideStorage, 'pdg-csv')); + sink.addNode({ + id: BB(1), + label: 'BasicBlock', + properties: { name: '', filePath: 'src/a.ts', startLine: 3, endLine: 4, text: 'm' }, + }); + const manifest = sink.finalize(); + + try { + await expect( + adapter.loadGraphToLbug(leakyGraph, collideBase, collideStorage, undefined, manifest), + ).rejects.toThrow(/collides with a structural node CSV for "BasicBlock"/); + } finally { + await fs.rm(collideBase, { recursive: true, force: true }); + } + }); +}); diff --git a/gitnexus/test/unit/lbug-native-safe-path.test.ts b/gitnexus/test/unit/lbug-native-safe-path.test.ts index 5d13b6a36..48d3c4711 100644 --- a/gitnexus/test/unit/lbug-native-safe-path.test.ts +++ b/gitnexus/test/unit/lbug-native-safe-path.test.ts @@ -5,7 +5,12 @@ * 8.3 short-name form before passing them to KuzuDB's native layer. */ import { describe, it, expect } from 'vitest'; -import { toNativeSafePath, cleanupNativePathJunctions } from '../../src/core/lbug/lbug-config.js'; +import path from 'path'; +import { + toNativeSafePath, + cleanupNativePathJunctions, + resolveNativeSafeStorageDir, +} from '../../src/core/lbug/lbug-config.js'; describe('toNativeSafePath', () => { it('returns ASCII paths unchanged on any platform', () => { @@ -61,3 +66,52 @@ describe('toNativeSafePath', () => { }); } }); + +describe('resolveNativeSafeStorageDir (#2202)', () => { + it('returns / for an ASCII storage path on any platform', () => { + const storage = path.join('repo', '.gitnexus'); + expect(resolveNativeSafeStorageDir(storage, 'csv')).toBe(path.join(storage, 'csv')); + expect(resolveNativeSafeStorageDir(storage, 'pdg-csv')).toBe(path.join(storage, 'pdg-csv')); + }); + + it('keeps csv and pdg-csv distinct (no collision between structural and streamed dirs)', () => { + const storage = path.join('repo', '.gitnexus'); + expect(resolveNativeSafeStorageDir(storage, 'csv')).not.toBe( + resolveNativeSafeStorageDir(storage, 'pdg-csv'), + ); + }); + + if (process.platform !== 'win32') { + it('does NOT relocate a non-ASCII storage path off Windows (platform gate)', () => { + const storage = path.join('repo', '用户', '.gitnexus'); + // Non-win32: the relocation never fires regardless of non-ASCII chars. + expect(resolveNativeSafeStorageDir(storage, 'pdg-csv')).toBe(path.join(storage, 'pdg-csv')); + }); + } + + if (process.platform === 'win32') { + it('relocates a non-ASCII storage path to a unique mkdtemp os.tmpdir() dir per subdir', () => { + const fs = require('fs'); + const storage = 'C:\\Project\\中文\\.gitnexus'; + // mkdtemp creates the dirs — track + clean them up. + const csv = resolveNativeSafeStorageDir(storage, 'csv'); + const pdg = resolveNativeSafeStorageDir(storage, 'pdg-csv'); + try { + // Both relocated under os.tmpdir(), ASCII-prefixed, and distinct (each + // mkdtemp call returns a fresh random suffix — never a predictable name). + expect(pdg.includes('gitnexus-pdg-csv-')).toBe(true); + expect(csv.includes('gitnexus-csv-')).toBe(true); + expect(csv).not.toBe(pdg); + // Two calls for the same (storage, subdir) yield DIFFERENT dirs (random). + const csv2 = resolveNativeSafeStorageDir(storage, 'csv'); + expect(csv2).not.toBe(csv); + // The relocated paths are not under the original non-ASCII storage path. + expect(pdg.includes('中文')).toBe(false); + fs.rmSync(csv2, { recursive: true, force: true }); + } finally { + fs.rmSync(csv, { recursive: true, force: true }); + fs.rmSync(pdg, { recursive: true, force: true }); + } + }); + } +}); diff --git a/gitnexus/test/unit/lbug/pdg-emit-sink.test.ts b/gitnexus/test/unit/lbug/pdg-emit-sink.test.ts new file mode 100644 index 000000000..7d4699529 --- /dev/null +++ b/gitnexus/test/unit/lbug/pdg-emit-sink.test.ts @@ -0,0 +1,331 @@ +/** + * PdgEmitSink unit tests (issue #2202 U2). + * + * Verifies the streaming PDG emit sink: + * - routes BasicBlock nodes + PDG edges to bounded CSV-on-disk; + * - delegates structural nodes/edges + the whole-program TAINT_PATH edge to + * the real graph (never streamed); + * - is byte-identical (set-wise) to the whole-graph `streamAllCSVsToDisk` + * emit for the same node/edge set (the issue's byte-identity acceptance); + * - never accumulates the PDG layer in the in-memory graph (the RSS bound). + */ +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'; +import fs from 'node:fs'; +import fsp from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; + +import { createKnowledgeGraph } from '../../../src/core/graph/graph.js'; +import { streamAllCSVsToDisk, buildBasicBlockRow } from '../../../src/core/lbug/csv-generator.js'; +import { PdgEmitSink } from '../../../src/core/lbug/pdg-emit-sink.js'; +import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; + +const bbNode = (fp: string, idx: number, line: number): GraphNode => ({ + id: `BasicBlock:${fp}:1:0:${idx}`, + label: 'BasicBlock', + properties: { name: '', filePath: fp, startLine: line, endLine: line + 1, text: `blk ${idx}` }, +}); + +const pdgEdge = ( + fp: string, + from: number, + to: number, + type: GraphRelationship['type'], + reason: string, +): GraphRelationship => ({ + id: `${type}:${fp}:${from}->${to}`, + sourceId: `BasicBlock:${fp}:1:0:${from}`, + targetId: `BasicBlock:${fp}:1:0:${to}`, + type, + confidence: 1, + reason, +}); + +/** Sorted non-empty lines of a CSV file (order-independent comparison). */ +const sortedLines = async (csvPath: string): Promise => { + const text = await fsp.readFile(csvPath, 'utf8'); + return text + .split('\n') + .filter((l) => l.length > 0) + .sort(); +}; + +let tmpRoot: string; + +beforeEach(() => { + tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'pdg-sink-')); +}); + +afterEach(() => { + fs.rmSync(tmpRoot, { recursive: true, force: true }); +}); + +describe('PdgEmitSink — routing', () => { + it('routes BasicBlock nodes and PDG edges to CSV, never to the real graph', () => { + const real = createKnowledgeGraph(); + const sink = new PdgEmitSink(real, path.join(tmpRoot, 'pdg-csv')); + + sink.addNode(bbNode('a.ts', 0, 1)); + sink.addNode(bbNode('a.ts', 1, 5)); + sink.addRelationship(pdgEdge('a.ts', 0, 1, 'CFG', 'seq')); + sink.addRelationship(pdgEdge('a.ts', 0, 1, 'REACHING_DEF', 'x:1:0')); + + // PDG layer must not land in the in-memory graph (the RSS bound). + expect(real.nodeCount).toBe(0); + expect(real.relationshipCount).toBe(0); + expect(sink.nodeCount).toBe(0); + + sink.finalize(); + }); + + it('delegates structural nodes, CALLS, and the whole-program TAINT_PATH edge to the real graph', () => { + const real = createKnowledgeGraph(); + const sink = new PdgEmitSink(real, path.join(tmpRoot, 'pdg-csv')); + + sink.addNode({ + id: 'Function:a.ts:fn:1', + label: 'Function', + properties: { name: 'fn', filePath: 'a.ts', startLine: 1, endLine: 9 }, + }); + sink.addRelationship({ + id: 'CALLS:1', + sourceId: 'Function:a.ts:fn:1', + targetId: 'Function:a.ts:fn2:9', + type: 'CALLS', + confidence: 1, + reason: '', + }); + // TAINT_PATH is a whole-program (Function→Function) edge — NOT streamed. + sink.addRelationship({ + id: 'TAINT_PATH:1', + sourceId: 'Function:a.ts:fn:1', + targetId: 'Function:a.ts:fn2:9', + type: 'TAINT_PATH', + confidence: 0.9, + reason: 'src->sink', + }); + + expect(real.nodeCount).toBe(1); + expect(real.relationshipCount).toBe(2); + expect(real.getNode('Function:a.ts:fn:1')).toBeDefined(); + + const manifest = sink.finalize(); + // No BasicBlock node CSV was created (no BasicBlock nodes were routed). + expect(manifest.nodeFiles.size).toBe(0); + expect(manifest.relsByPair.size).toBe(0); + }); +}); + +describe('PdgEmitSink — byte-identity vs whole-graph emit', () => { + it('streamed CSV line set equals streamAllCSVsToDisk for the same nodes/edges', async () => { + const fp = 'a.ts'; + const nodes = [bbNode(fp, 0, 1), bbNode(fp, 1, 5), bbNode(fp, 2, 9)]; + const edges: GraphRelationship[] = [ + pdgEdge(fp, 0, 1, 'CFG', 'seq'), + pdgEdge(fp, 1, 2, 'CFG', 'cond-true'), + pdgEdge(fp, 0, 2, 'REACHING_DEF', 'x:1:0'), + pdgEdge(fp, 1, 2, 'CDG', 'T'), + pdgEdge(fp, 0, 1, 'POST_DOMINATE', ''), + pdgEdge(fp, 0, 2, 'TAINTED', 'taint'), + pdgEdge(fp, 1, 2, 'SANITIZES', 'clean'), + ]; + + // Whole-graph path: add to a plain graph, run streamAllCSVsToDisk. + const wholeGraph = createKnowledgeGraph(); + for (const n of nodes) wholeGraph.addNode(n); + for (const e of edges) wholeGraph.addRelationship(e); + const wholeDir = path.join(tmpRoot, 'csv'); + await streamAllCSVsToDisk(wholeGraph, path.join(tmpRoot, 'no-such-repo'), wholeDir); + + // Streamed path: route the same set through the sink. + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + for (const n of nodes) sink.addNode(n); + for (const e of edges) sink.addRelationship(e); + const manifest = sink.finalize(); + + // BasicBlock node CSV: identical line set. + expect(await sortedLines(path.join(pdgDir, 'basicblock.csv'))).toEqual( + await sortedLines(path.join(wholeDir, 'basicblock.csv')), + ); + + // PDG edges all route to the BasicBlock|BasicBlock pair file: identical set. + expect(await sortedLines(path.join(pdgDir, 'rel_BasicBlock_BasicBlock.csv'))).toEqual( + await sortedLines(path.join(wholeDir, 'rel_BasicBlock_BasicBlock.csv')), + ); + + // Manifest reports the streamed files + row counts. + expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(nodes.length); + expect(manifest.relsByPair.get('BasicBlock|BasicBlock')?.rows).toBe(edges.length); + }); + + it('emits rows via the shared builder (buildBasicBlockRow)', async () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + const n = bbNode('a.ts', 0, 3); + sink.addNode(n); + sink.finalize(); + const lines = await sortedLines(path.join(pdgDir, 'basicblock.csv')); + // header + one data row; the data row is exactly buildBasicBlockRow(n). + expect(lines).toContain(buildBasicBlockRow(n)); + }); +}); + +describe('PdgEmitSink — bounded retention', () => { + it('flushes incrementally so the graph never holds the PDG layer', async () => { + const real = createKnowledgeGraph(); + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const CHUNK = 2; + const sink = new PdgEmitSink(real, pdgDir, CHUNK); // tiny chunk to force flushes + + // TOTAL is intentionally NOT a multiple of CHUNK so the final partial chunk + // is genuinely still buffered (unflushed) at the mid-stream read. With a + // multiple (e.g. 50 % 2 === 0) the last addRow's flush would have written + // every row and the "mid-stream" assertion would prove nothing (#2202 + // review #7). + const TOTAL = 51; + const REMAINDER = TOTAL % CHUNK; // 1 — must be non-zero + expect(REMAINDER).toBeGreaterThan(0); + for (let i = 0; i < TOTAL; i++) sink.addNode(bbNode('a.ts', i, i)); + + // Mid-stream (before finalize): exactly the whole flushed chunks are on + // disk; the partial last chunk (REMAINDER rows) is still buffered in memory, + // proving the writer streams to the OS and never buffers the whole layer. + const midText = fs.readFileSync(path.join(pdgDir, 'basicblock.csv'), 'utf8'); + const midDataRows = midText.split('\n').filter((l) => l.length > 0).length - 1; // minus header + expect(midDataRows).toBe(TOTAL - REMAINDER); // 50 flushed, 1 still buffered + expect(TOTAL - midDataRows).toBe(REMAINDER); // exactly the unflushed remainder + expect(TOTAL - midDataRows).toBeLessThanOrEqual(CHUNK); // unflushed is bounded by one chunk + + // The in-memory graph never received a single BasicBlock. + expect(real.nodeCount).toBe(0); + + const manifest = sink.finalize(); + expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(TOTAL); + const finalRows = (await sortedLines(path.join(pdgDir, 'basicblock.csv'))).length - 1; + expect(finalRows).toBe(TOTAL); // finalize flushed the buffered remainder + }); + + it('finalize twice throws', () => { + const sink = new PdgEmitSink(createKnowledgeGraph(), path.join(tmpRoot, 'pdg-csv')); + sink.finalize(); + expect(() => sink.finalize()).toThrow(/twice/); + }); +}); + +describe('PdgEmitSink — pass-through contract (dedup is the caller’s)', () => { + // The sink does NOT dedup by id (that would retain every id → O(total ids) + // memory, undermining the O(chunk) bound). Cross-pass dedup is done upstream, + // per file, in run.ts (a file imported by two language passes is emitted + // once). The sink is a faithful pass-through: it writes every id it is given + // and must not be fed duplicates. See #2202 finding #1 + the run-loop / Vue+TS + // integration coverage for the cross-pass dedup itself. + it('writes every BasicBlock it is given (no id dedup in the sink)', async () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + const n = bbNode('a.ts', 0, 1); + sink.addNode(n); + sink.addNode(n); // sink does not dedup — both rows are written + const manifest = sink.finalize(); + expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(2); + expect((await sortedLines(path.join(pdgDir, 'basicblock.csv'))).length - 1).toBe(2); + }); + + it('writes every PDG edge it is given (no id dedup in the sink)', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + const e = pdgEdge('a.ts', 0, 1, 'CFG', 'seq'); + sink.addRelationship(e); + sink.addRelationship(e); + const manifest = sink.finalize(); + expect(manifest.relsByPair.get('BasicBlock|BasicBlock')?.rows).toBe(2); + }); + + it('skips a PDG edge whose endpoint label is not a node table', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + // sourceId prefix "Bogus" is not in NODE_TABLES → skipped (mirrors RelPairRouter). + sink.addRelationship({ + id: 'CFG:bogus', + sourceId: 'Bogus:a.ts:0', + targetId: 'BasicBlock:a.ts:1:0:1', + type: 'CFG', + confidence: 1, + reason: 'seq', + }); + const manifest = sink.finalize(); + expect(manifest.relsByPair.size).toBe(0); + }); +}); + +describe('PdgEmitSink — IO failure poisoning (#2202 review #4/#6)', () => { + // A streamed-write failure is an IO fault, not the CFG-logic error the emit + // loop's per-file try/catch is built to swallow. The sink poisons the failing + // writer (or records an open failure) so finalize fails loudly instead of + // returning a truncated manifest that the bulk COPY would silently load. + + afterEach(() => { + vi.restoreAllMocks(); + }); + + it('poisons the writer on a mid-stream write failure (disk-full) → finalize throws', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1); // flush every row + + // The writer opens fine (real openSync); the flush's writeSync fails. + const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => { + throw new Error('ENOSPC: no space left on device'); + }); + // The throw propagates to the immediate caller (the emit loop, which would + // swallow it as a per-file CFG error) — that is exactly why finalize must + // re-check poison below. + expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/ENOSPC/); + spy.mockRestore(); + + expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/); + }); + + it('records an openSync failure (EMFILE) → finalize throws even if the caller swallowed it', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); // ctor mkdir/rm run before the spy + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + + const spy = vi.spyOn(fs, 'openSync').mockImplementation(() => { + throw new Error('EMFILE: too many open files'); + }); + expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/EMFILE/); + spy.mockRestore(); + + expect(() => sink.finalize()).toThrow(/IO error|EMFILE/); + }); + + it('surfaces a final-flush IO failure from finalize (rows buffered, never mid-flushed)', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1000); // big chunk → no mid flush + sink.addNode(bbNode('a.ts', 0, 1)); // buffered only (1 < 1000) + + // The only writeSync happens during the final flush inside finalize → close. + const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => { + throw new Error('ENOSPC: disk full at close'); + }); + // close() never throws (it records poison); finalize reports it. + expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/); + spy.mockRestore(); + }); + + it('a poisoned writer stops accepting rows (no unbounded buffering on a dead fd)', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1); + + const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => { + throw new Error('ENOSPC'); + }); + expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/ENOSPC/); // poisons the writer + // Subsequent rows are dropped silently at the writer (it is dead) — they do + // not re-throw and do not accumulate; the run still fails at finalize. + expect(() => sink.addNode(bbNode('a.ts', 1, 2))).not.toThrow(); + expect(() => sink.addNode(bbNode('a.ts', 2, 3))).not.toThrow(); + spy.mockRestore(); + + expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/); + }); +}); diff --git a/gitnexus/test/unit/rel-pair-routing.test.ts b/gitnexus/test/unit/rel-pair-routing.test.ts new file mode 100644 index 000000000..8355014d7 --- /dev/null +++ b/gitnexus/test/unit/rel-pair-routing.test.ts @@ -0,0 +1,176 @@ +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { EventEmitter } from 'events'; +import fs from 'fs'; +import path from 'path'; +import os from 'os'; +import { RelPairRouter, getNodeLabel } from '../../src/core/lbug/rel-pair-routing.js'; + +/** + * Unit tests for RelPairRouter (#2203 U2) — the production per-pair emit path. + * + * Mirrors test/unit/rel-csv-split.test.ts: drives the router with an injected + * mock WriteStream factory so the error, backpressure, and teardown paths are + * exercised without LadybugDB or real disk streams. These paths are otherwise + * unreachable in the integration suite (which only hits the no-backpressure + * happy path), so this is the coverage for the router's failure modes. + */ + +// Controllable backpressure + error injection (same shape as the split oracle's mock). +class MockWriteStream extends EventEmitter { + public chunks: string[] = []; + public destroyed = false; + public ended = false; + public blocked = false; + public maxDrainListenersSeen = 0; + // State flags + events so `stream/promises.finished(ws)` (used by the + // router's close()) resolves against this mock instead of hanging. + public writable = true; + public writableEnded = false; + public writableFinished = false; + + write(chunk: string): boolean { + this.chunks.push(chunk); + const count = this.listenerCount('drain'); + if (count > this.maxDrainListenersSeen) this.maxDrainListenersSeen = count; + return !this.blocked; + } + + end(cb?: (err?: Error) => void): this { + this.ended = true; + this.writableEnded = true; + this.writableFinished = true; + this.writable = false; + if (cb) cb(); + queueMicrotask(() => { + this.emit('finish'); + this.emit('close'); + }); + return this; + } + + destroy(): this { + this.destroyed = true; + return this; + } + + unblock(): void { + this.blocked = false; + this.emit('drain'); + } + + triggerError(err: Error): void { + this.emit('error', err); + } +} + +const HEADER = '"from","to","type","confidence","reason","step"'; +const VALID = new Set(['File', 'Function', 'Community', 'Process']); + +const row = (from: string, to: string, type = 'CALLS'): string => + `"${from}","${to}","${type}",1.0,"auto",0`; + +function mockFactory(streams: MockWriteStream[], opts?: { blocked?: boolean }) { + return (() => { + const ws = new MockWriteStream(); + if (opts?.blocked) ws.blocked = true; + streams.push(ws); + return ws; + }) as unknown as (filePath: string) => import('fs').WriteStream; +} + +let tmpDir: string; + +beforeEach(() => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'rel-pair-routing-test-')); +}); + +afterEach(() => { + fs.rmSync(tmpDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 50 }); +}); + +describe('getNodeLabel', () => { + it('maps comm_/proc_ prefixes and otherwise splits on the first colon', () => { + expect(getNodeLabel('comm_42')).toBe('Community'); + expect(getNodeLabel('proc_7')).toBe('Process'); + expect(getNodeLabel('Function:src/a.ts:f:1')).toBe('Function'); + expect(getNodeLabel('File:src/a.ts')).toBe('File'); + }); +}); + +describe('RelPairRouter', () => { + it('routes valid edges to per-pair files (header first) and skips invalid-label edges', async () => { + const streams: MockWriteStream[] = []; + const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams)); + + const route = async (from: string, to: string) => { + const p = router.route(from, to, row(from, to)); + if (p) await p; + }; + await route('File:a', 'Function:a:f:1'); + await route('File:a', 'Function:a:g:2'); // same pair + await route('Function:a:f:1', 'Function:a:g:2'); // different pair + await route('Bogus:x', 'File:a'); // invalid FROM label → skipped + await route('File:a', 'Bogus:y'); // invalid TO label → skipped (other branch) + await router.close(); + + expect(router.skipped).toBe(2); + expect(router.total).toBe(3); + expect([...router.byPair.keys()].sort()).toEqual(['File|Function', 'Function|Function']); + expect(router.byPair.get('File|Function')!.rows).toBe(2); + // Header is the first chunk written to each pair stream. + expect(streams[0].chunks[0]).toBe(HEADER + '\n'); + expect(streams.every((s) => s.ended)).toBe(true); + }); + + it('returns a drain promise under backpressure and completes once unblocked', async () => { + const streams: MockWriteStream[] = []; + const router = new RelPairRouter( + tmpDir, + HEADER, + VALID, + mockFactory(streams, { blocked: true }), + ); + + const pending = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1')); + expect(pending).toBeInstanceOf(Promise); // header write hit backpressure + streams[0].unblock(); + await pending; + + expect(streams[0].maxDrainListenersSeen).toBeLessThanOrEqual(1); + expect(streams[0].chunks[0]).toBe(HEADER + '\n'); + expect(router.total).toBe(1); + }); + + it('on a stream error: route() throws the real error, lastError exposes it, close() rejects + destroys', async () => { + const streams: MockWriteStream[] = []; + const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams)); + + const first = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1')); + if (first) await first; + + const err = new Error('EMFILE: too many open files'); + streams[0].triggerError(err); + + // The next route surfaces the REAL error, not a generic AbortError. + expect(() => router.route('File:a', 'Function:a:g:2', row('File:a', 'Function:a:g:2'))).toThrow( + 'EMFILE', + ); + expect(router.lastError).toBe(err); + await expect(router.close()).rejects.toThrow('EMFILE'); + expect(streams[0].destroyed).toBe(true); + }); + + it('destroy() tears down every open pair stream', async () => { + const streams: MockWriteStream[] = []; + const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams)); + + const a = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1')); + if (a) await a; + const b = router.route('Community:1', 'Community:2', row('Community:1', 'Community:2')); + if (b) await b; + + router.destroy(); + expect(streams.length).toBe(2); + expect(streams.every((s) => s.destroyed)).toBe(true); + }); +}); diff --git a/gitnexus/test/unit/stream-pdg-emit-config.test.ts b/gitnexus/test/unit/stream-pdg-emit-config.test.ts new file mode 100644 index 000000000..9d45e8f8a --- /dev/null +++ b/gitnexus/test/unit/stream-pdg-emit-config.test.ts @@ -0,0 +1,99 @@ +/** + * Streaming PDG-emit config gating (issue #2202 U3). + * + * Verifies: + * - `resolveStreamPdgEmit` engages only when pdg + full-rebuild (force) + an + * enable signal (explicit option OR GITNEXUS_STREAM_PDG_EMIT) all hold; + * - `resolvePdgEmitChunkSize` prefers the explicit option, falls back to + * GITNEXUS_PDG_EMIT_CHUNK_SIZE, else undefined; + * - the memory-only streaming knobs are NOT stamped into RepoMeta.pdg, so + * changing them never trips `pdgModeMismatch` (would force needless full + * writebacks otherwise). + */ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import { + resolveStreamPdgEmit, + resolvePdgEmitChunkSize, + pdgModeMismatch, + resolvePdgConfig, +} from '../../src/core/run-analyze.js'; + +afterEach(() => { + vi.unstubAllEnvs(); +}); + +describe('resolveStreamPdgEmit — gating', () => { + it('engages only with pdg + force + an enable signal', () => { + // explicit option + expect(resolveStreamPdgEmit({ pdg: true, force: true, streamPdgEmit: true })).toBe(true); + // env toggle + vi.stubEnv('GITNEXUS_STREAM_PDG_EMIT', '1'); + expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(true); + }); + + it('does NOT engage without --force (incremental writeback reads BasicBlocks back)', () => { + expect(resolveStreamPdgEmit({ pdg: true, force: false, streamPdgEmit: true })).toBe(false); + expect(resolveStreamPdgEmit({ pdg: true, streamPdgEmit: true })).toBe(false); + }); + + it('does NOT engage without --pdg (nothing to stream)', () => { + expect(resolveStreamPdgEmit({ pdg: false, force: true, streamPdgEmit: true })).toBe(false); + }); + + it('does NOT engage with no enable signal', () => { + expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(false); + vi.stubEnv('GITNEXUS_STREAM_PDG_EMIT', '0'); + expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(false); + }); +}); + +describe('resolvePdgEmitChunkSize', () => { + it('prefers the explicit option over the env var', () => { + vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '128'); + expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 999 })).toBe(999); + }); + + it('falls back to the env var, then to undefined', () => { + vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '128'); + expect(resolvePdgEmitChunkSize({})).toBe(128); + vi.unstubAllEnvs(); + expect(resolvePdgEmitChunkSize({})).toBeUndefined(); + }); + + it('rejects non-positive / non-integer env values (falls back to undefined)', () => { + for (const bad of ['0', '-5', 'abc', '1.5', '']) { + vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', bad); + expect(resolvePdgEmitChunkSize({})).toBeUndefined(); + } + }); + + it('rejects an explicit non-positive option (0/negative is not nullish — would defeat buffering)', () => { + expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 0 })).toBeUndefined(); + expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: -10 })).toBeUndefined(); + // ...but an explicit 0 still falls back to a valid env value when present. + vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '256'); + expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 0 })).toBe(256); + }); +}); + +describe('streaming knobs are NOT emit-affecting (no pdgModeMismatch)', () => { + it('changing streamPdgEmit / pdgEmitChunkSize does not trip pdgModeMismatch', () => { + const base = { pdg: true as const }; + const recorded = resolvePdgConfig(base); + // Same pdg config, but with the streaming knobs flipped — must NOT mismatch. + expect( + pdgModeMismatch(recorded, { + ...base, + streamPdgEmit: true, + pdgEmitChunkSize: 64, + } as Parameters[1]), + ).toBe(false); + }); + + it('streaming knobs are absent from the resolved RepoMeta.pdg stamp', () => { + const stamp = resolvePdgConfig({ pdg: true }); + expect(stamp).toBeDefined(); + expect(stamp).not.toHaveProperty('streamPdgEmit'); + expect(stamp).not.toHaveProperty('pdgEmitChunkSize'); + }); +}); diff --git a/package-lock.json b/package-lock.json index 65c0870ed..0969cc7a7 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1868,10 +1868,20 @@ "license": "MIT" }, "node_modules/js-yaml": { - "version": "4.1.1", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.1.1.tgz", - "integrity": "sha512-qQKT4zQxXl8lLwBtHMWwaTcGfFOZviOJet3Oy/xmGk2gZH677CJM9EvtfdSkgWcATZhj/55JZ0rmy3myCT5lsA==", + "version": "4.2.0", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.2.0.tgz", + "integrity": "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw==", "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/puzrin" + }, + { + "type": "github", + "url": "https://github.com/sponsors/nodeca" + } + ], "license": "MIT", "dependencies": { "argparse": "^2.0.1"