From 6932e7a9fdfa4fe56eba1001c3c3aa159df8d267 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 15 Jun 2026 12:31:04 +0100 Subject: [PATCH 01/26] feat(cfg): PDG/CFG visitors for all supported languages (#2195) (#2197) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * test(cfg): validate cfg/visitors literals + drop 3 dead TS node types Extend the grammar-literal CI gate (test/helpers/literal-collectors.ts) to scan cfg/visitors/*.ts, mapping each visitor file to its grammar via the existing basename rule (c-cpp -> C/C++, csharp -> C#, java -> Java, go -> Go, typescript -> TS). Closes the gap where the gate never validated CFG visitor node-type literals -- the prerequisite for adding C-family visitors safely (#2195 U1). The newly-scanned TS visitor surfaced 3 dead literals absent from every grammar it serves (typescript/javascript/tsx all = 0): for_of_statement (for-of parses as for_in_statement), async_function_declaration and async_arrow_function (async functions are function_declaration / arrow_function + an async child). Removed them; behavior-preserving -- the cases never matched, bench --check fingerprints unchanged, TS visitor unit tests green. Co-Authored-By: Claude Opus 4.8 (1M context) * test(cfg): language-agnostic CFG unit-test harness (#2195 U1) Extract the grammar-agnostic engine from ts-cfg-harness into makeCfgHarness(grammar, visitor, filePath) at test/helpers/cfg-harness.ts. Function discovery delegates to visitor.isFunction, so the harness carries no language-specific node-type knowledge -- each C-family visitor's unit tests can drive the real worker-side builder against real source. ts-cfg-harness becomes a thin TS binding re-exporting the same parse/collectFunctions/cfgOf/cfgsOf (behavior-preserving: all 5 existing consumers -- taint propagate/model-match/summary-harvest/taint-emit + cfg harvest -- pass unchanged, 223 tests green). New harness.test.ts proves TS-faithfulness and isFunction-delegation via a stub visitor. The bench parameterization (measure.mjs) is sequenced into U7, where the first C-family scaling scenario makes the {grammar, visitorFactory} seam validatable against a real non-TS language. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): C and C++ CFG visitor + def/use harvest (#2195 U2) Add createCCfgVisitor/createCppCfgVisitor over a shared CCfgWalk core. Grammar introspection confirmed tree-sitter-c and tree-sitter-cpp share every control-flow node type/field, so CppCfgWalk extends CCfgWalk with only the C++-only nodes (try/catch/throw/for_range_loop/lambda) via a visitExtra hook -- no language conditionals (AGENTS no-language-naming). Wire both into c-cpp.ts providers. Harvest (c-cpp-harvest.ts): two-phase binding table + per-statement defs/uses/mayDefs (no sites[] yet -- U6). Edge kinds match the TS contract; functionStartColumn populated; non-terminating loops (for(;;), while(1)) emit the structural exit-escape edge so EXIT stays reverse-reachable and CDG is not silently skipped -- verified against the production post-dominator + control-dependence solvers (for(;;) -> 3 CDG edges). buildFunctionCfg returns undefined rather than throwing. 23 real-parser regression tests; grammar-literal gate green (literals validated against both grammars). Documented gaps: C++ RAII destructors, setjmp/longjmp, computed goto (route to EXIT + warn). Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): C# CFG visitor + def/use harvest (#2195 U3) Add createCsharpCfgVisitor + csharp-harvest over the shared CfgBuilder / ControlFlowContext, modeling the C# statement taxonomy: if/else, for/foreach/while/do, switch_section (+ switch_expression arms), try/catch/catch_filter/finally, using + lock (deterministic finalizers -- dispose/release runs on normal AND exception exit, finally-* completion edges on crossing jumps), goto/labeled, yield (surface only), return/ throw/break/continue. Wire into csharpProvider. Every literal validated against tree-sitter-c-sharp via the introspection probe (record_declaration, no else_clause, switch_section, positional access where no field exists). Edge kinds match the contract; functionStartColumn populated; while(true) keeps EXIT reverse-reachable (production CDG probe: 3 edges). buildFunctionCfg returns undefined rather than throwing. 34 real-parser regression tests; grammar-literal gate green; no regression (cfg unit dir 256/256, tsc clean). Documented gaps: yield iterator state machine, goto case/default, async suspension points. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): Java CFG visitor + def/use harvest (#2195 U4) Add createJavaCfgVisitor + java-harvest over the shared CfgBuilder / ControlFlowContext: if/else, classic for, enhanced-for, while, do-while, classic-vs-arrow switch (switch_block_statement_group fallthrough vs switch_rule no-fallthrough), try/catch/finally + try-with-resources (auto-close synthesized as a finalizer, closes on normal AND exception exit) + synchronized (monitor-release finalizer), labeled break/continue to the labeled frame, yield, return/throw/break/continue. Wire into javaProvider. Every literal validated against tree-sitter-java via the probe (switch_expression covers both switch forms, generic_type, line_comment, for init field). Edge kinds match the contract; functionStartColumn populated; while(true)/for(;;) keep EXIT reverse-reachable (production CDG probe: 3 edges; hazard fixture: 34 CDG edges). buildFunctionCfg returns undefined rather than throwing. 43 real-parser regression tests; grammar-literal gate green; no regression (cfg unit suite 304, tsc clean). Documented gaps: switch-as- expression-value inline, yield state machine, async/field-write defs. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): Go CFG visitor + def/use harvest (#2195 U5) Add createGoCfgVisitor + go-harvest, the highest-divergence target: for_statement (all four shapes -- for_clause C-style, while-style, range_clause, bare for{}), expression/type switch (no implicit fallthrough) + explicit fallthrough_statement, select_statement, defer (LIFO finalizer legs at function exit), go (call is straight-line; the closure body is its own CFG via isFunction), labeled break/continue/goto, multiple-return assigns (a, b := f() defines each LHS). Wire into goProvider. CRITICAL (review A2): every non-terminating shape -- for{}, for cond{}, select{} with no default -- emits a structural exit-escape edge so EXIT stays reverse-reachable and the production CDG is not silently skipped. Verified: for{} -> CDG=3, select{} -> CDG=1, for-range -> CDG=2, all exitReachable=true. Every literal validated against tree-sitter-go via the probe. 32 real-parser regression tests; grammar-literal gate green; no regression (186 across all 5 visitors + gate, full cfg unit 331, tsc clean). Documented gaps: panic/recover unwind, goroutine happens-before. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): call-site sites[] taint substrate for C-family (#2195 U6) Extend the C/C++/C#/Java/Go harvests with the call-site sites[] taint substrate (SiteRecord/SiteArgOccurrence), mirroring the TS shape so the shared taint matcher consumes all languages uniformly. Extract the grammar-agnostic site machinery into cfg/visitors/call-site-harvest.ts (CallSiteFactAccumulator -- names no language); each harvest adds only its per-grammar visitCall/walkChain over its call node (C/C++ call_expression, C# invocation_expression, Java method_invocation, Go call_expression). INERT BY DESIGN: no C-family taint model exists (registerBuiltinTaintModels is TS/JS only), so getSourceSinkConfig returns undefined for these languages and the harvested sites produce ZERO TAINTED edges -- the positive source->sink->TAINTED path is deferred with the model authoring. sites emitted only when non-empty; facts-only attachment, block/edge topology unchanged (pre-existing topology + def/use tests byte-identical). 23 new substrate tests; 574 green across the cfg/taint/emit suites; gate green; tsc clean. Co-Authored-By: Claude Opus 4.8 (1M context) * test(cfg): worker-mode PDG integration + bench parameterization (#2195 U7) Prove the five C-family visitors build PDG through the REAL worker pipeline. pipeline-pdg.test.ts: per-language (C/C++/C#/Java/Go) temp repo run with pdg:true asserts BasicBlock+CFG+REACHING_DEF+CDG all > 0 (CDG>0 proves EXIT stays reverse-reachable end-to-end through the worker, incl. each fixture's non-terminating loop/select); a paired run with pdg off asserts == 0, the two flag-off graphs byte-identical (R3), no PDG types leak, pinned by a golden snapshot. Counts e.g. Go 151 BB / 56 CDG. Parameterize bench/cfg/measure.mjs by a per-language LANGS registry resolved generically via getLanguageGrammar + getProvider(X).cfgVisitor (no static import table). Default TS byte-identical -- all 6 TS fingerprints unchanged under --check; taint-dense stays TS-only (TS_JS_TAINT_MODEL never runs against model-less C-family CFGs). Add a go:branchy scenario+baseline (namespaced) -- its fingerprint shape (32 blocks/46 edges) matches TS branchy, cross-validating the Go visitor. 15 pipeline tests + bench --check PASS (7 scenarios); 354 unit cfg green; dist rebuilt clean. Absorbs the bench parameterization deferred from U1. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): Python CFG visitor + def/use harvest (#2195 U8) Add createPythonCfgVisitor + python-harvest -- the most structurally divergent target (indentation blocks, elif, for/while-else, with, try/ except/except-group/else/finally, match/case, comprehensions, walrus), confirming the shared CfgBuilder/ControlFlowContext core carries no brace-family assumptions. for/while else-clause sits on the normal- completion edge (not break); with modeled as try/finally dispose; match has no fallthrough. Wire into pythonProvider. Every literal validated against tree-sitter-python via the probe. while True: keeps EXIT reverse-reachable (production CDG probe: 3 edges; fixture: 42 CDG edges). 37 real-parser tests; gate green; no regression (cfg unit 391, tsc clean). Gaps: async/generator suspension, comprehension scope over-approximation. No sites[] (taint substrate, separate). Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): PHP CFG visitor + def/use harvest (#2195 U9) Add createPhpCfgVisitor + php-harvest: if/elseif/else (+ alt colon syntax), for/foreach/while/do-while, switch (fallthrough) + match (no fallthrough), try/catch/finally, break N/continue N (N-th enclosing loop), goto, return/throw. Wire into phpProvider. Every literal validated against tree-sitter-php (php_only) via the probe (for_statement initialize/condition/update; throw_expression not throw_statement; break/continue integer child). while(true) keeps EXIT reverse-reachable (production CDG probe: 3 edges; break 2 escapes the outer loop). 35 real-parser tests. Also repoint worker-roundtrip's "non-CFG language" gate test from Python (which now has a cfgVisitor) to COBOL (the permanent non-goal of the rollout) -- a stale assertion the Python commit invalidated. Full in-process sweep green (452 across 18 files). Gaps: match inline value, goto plain-block. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): Ruby CFG visitor + def/use harvest (#2195 U10) Add createRubyCfgVisitor + ruby-harvest: if/unless/elsif/else + statement-modifier forms (x if c, x while c), while/until/for (until inverts the sense), case/when + case/in (pattern, no fallthrough), begin/rescue/else/ensure (ensure=finally, rescue=catch) + retry (loop-back into begin), return/break/next/redo, blocks/lambdas as their own closure CFGs. Wire into rubyProvider. Every literal validated against tree-sitter-ruby via the probe (case vs case_match, modifier nodes, typed rescue/ensure children). loop do / while true keep EXIT reverse-reachable (production CDG probe: 3 edges). 34 real-parser tests; comprehensive sweep green (486). Gaps: yield, expression-position if/case/begin inline, ivar/gvar non-local defs. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): Rust CFG visitor + def/use harvest (#2195 U11) Add createRustCfgVisitor + rust-harvest for the expression-oriented Rust: if/else + if-let, loop (infinite -- structural escape edge), while/ while-let/for, match (no fallthrough) + guards, labeled break/continue ('outer), break-with-value, ? operator (try_expression) as an early-return throw edge to EXIT, let-else (diverging else). visitLet handles control-flow in value position (let x = loop/if/match). Wire into rustProvider. Every literal validated against tree-sitter-rust via the probe (label is a named child not a field; line_comment; _ pattern). loop {} keeps EXIT reverse-reachable (production CDG probe: 3 edges). 33 real-parser tests; comprehensive sweep green (519). Gaps: panic, async/.await, macro bodies. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): Swift CFG visitor + def/use harvest (#2195 U12) Add createSwiftCfgVisitor + swift-harvest (vendored tree-sitter-swift via requireVendoredGrammar): if/else + optional binding (if let), guard...else (diverging early exit), for-in/while/repeat-while (bottom-test), switch (no implicit fallthrough; explicit fallthrough keyword; where guards), do/catch + try/try?/try!, defer (LIFO finalizer at scope exit), labeled break/continue, control_transfer_statement (one node for break/continue/ return/throw). Wire into swiftProvider. Every literal validated against the vendored grammar via the probe (no block node; if-let folds into condition+bound_identifier; defer parses as a call_expression with trailing closure). while true keeps EXIT reverse-reachable (production CDG probe: 3 edges). 24 real-parser tests; comprehensive sweep green (543). Gaps: computed properties, defer block-scope approx, fatalError traps. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): Kotlin CFG visitor + def/use harvest (#2195 U13) Add createKotlinCfgVisitor + kotlin-harvest (vendored tree-sitter-kotlin): if/else, when (subject + subjectless, no fallthrough), for/while/do-while, try/catch/finally, jump_expression (return/return@/break/break@/continue/ continue@/throw), labeled loops, control_structure_body unwrapping, expression-body functions. The grammar is field-less for control flow, so the visitor navigates by child type+position. Wire into kotlinProvider. Every literal validated against the vendored grammar via the probe (line_comment/multiline_comment, not comment). while (true) keeps EXIT reverse-reachable (production CDG probe: 3 edges; worker-mode fixture: BB=82, CDG=41). 28 real-parser tests; comprehensive sweep green (571). Gaps: value-position if/when/try inline, inline-fun non-local return, getters/setters. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): Dart CFG visitor + def/use harvest (#2195 U14) Add createDartCfgVisitor + dart-harvest (vendored tree-sitter-dart): if/else, C-for/for-in/while/do-while, switch (empty-case fallthrough + explicit continue-label) + switch_expression, try/on/catch/finally + rethrow + assert (throw edges), return/break/continue/throw, labeled loops, arrow bodies, closures. Dart splits a function into sibling signature + function_body nodes, so the body (or function_expression) is the CFG-bearing node. Wire into dartProvider. Every literal validated against the vendored grammar via the probe (only constant_pattern exists; removed speculative relational/logical pattern names). while (true) keeps EXIT reverse-reachable (production CDG probe: 3 edges). 34 real-parser tests; comprehensive sweep green (605). Gaps: labeled-loop grammar quirk (read via ERROR sibling), async straight-line, value-position if/switch inline. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): Vue (reuse TS visitor) + worker-mode proof for all langs (#2195 U15) Vue SFC diff --git a/gitnexus/test/integration/cfg/pipeline-pdg.test.ts b/gitnexus/test/integration/cfg/pipeline-pdg.test.ts index 2177c761a..00f4ed8f7 100644 --- a/gitnexus/test/integration/cfg/pipeline-pdg.test.ts +++ b/gitnexus/test/integration/cfg/pipeline-pdg.test.ts @@ -2,10 +2,13 @@ import { describe, it, expect, afterAll } from 'vitest'; import fs from 'fs'; import os from 'os'; import path from 'path'; +import crypto from 'crypto'; import { runPipelineFromRepo } from '../../../src/core/ingestion/pipeline.js'; import type { PipelineResult } from '../../../src/types/pipeline.js'; import { decodeTaintPath } from '../../../src/core/ingestion/taint/path-codec.js'; import { fixtureTaintTotals } from '../../helpers/taint-fixture.js'; +import { isLanguageAvailable } from '../../../src/core/tree-sitter/parser-loader.js'; +import { SupportedLanguages } from '../../../src/config/supported-languages.js'; // U7 — end-to-end proof that the `--pdg` opt-in reaches BOTH sinks: the parse // worker builds a per-function CFG (workerData.pdg) and scope-resolution emits @@ -187,3 +190,356 @@ describe('U7 — end-to-end --pdg pipeline', () => { expect(cdg).toBe(0); }, 60000); }); + +// ── C-family worker-mode PDG (#2195 U7) ───────────────────────────────────── +// +// The same both-sinks proof as the TS block above, run through the REAL worker +// pipeline for each of C, C++, C#, Java, Go. Each language gets its own tiny +// repo (one hazard fixture with real branching AND a non-terminating +// loop/`select`) and we assert, under `--pdg`: +// - BasicBlock + CFG > 0 (the worker built a per-function CFG and emit wired it) +// - REACHING_DEF > 0 (the def/use harvest populates the data-dependence layer) +// - CDG > 0 AND ≥1 CDG edge is sourced INSIDE the non-terminating-loop +// function itself (`hazard`, below) — not merely an aggregate satisfied by +// any branching function in the fixture. This is the load-bearing claim: +// the post-dom/CDG pass was NOT skipped for the function whose loop traps +// EXIT, i.e. EXIT stays reverse-reachable end-to-end through the worker even +// with the non-terminating loop/`select`. (#2197 U3 — the prior whole- +// fixture `cdg > 0` aggregate did not isolate the hazard function.) +// and without `--pdg` (both the default run and an explicit `pdg:false` run): +// - BasicBlock + CFG + REACHING_DEF + CDG == 0 +// - the non-PDG graph is byte-identical between the two flag-off runs and +// matches a committed digest snapshot (the per-language byte-identical-off +// golden parity gate — R3; the cross-repo gate is pipeline-graph-golden). +// +// ⚠ Requires a FRESH `dist/parse-worker.js` — CFGs are built in the worker from +// `dist/`. A stale bundle silently zeros CFG output. `pretest:integration` (and +// the U7 verification recipe) run `node scripts/build.js` first. + +const C_FAMILY_FIXTURES = path.join(__dirname, 'fixtures'); + +// `hazard`: a substring of a BasicBlock's `text` that appears ONLY inside the +// fixture's non-terminating-loop function (`for(;;)` / `while(true)` / `for{}`). +// It locates that function's block anchor so the CDG assertion can prove the +// function specifically is CDG-bearing (see `cdgSourcedInHazardFunction`). C# +// has no such loop (its `Retry` goto-cycle is conditional and terminates), so +// it has no `hazard` and keeps the whole-fixture aggregate only. +const C_FAMILY: ReadonlyArray<{ lang: string; fixture: string; hazard?: string }> = [ + { lang: 'C', fixture: 'c-hazards.c', hazard: 'handle_request' }, // server_forever: for(;;) + { lang: 'C++', fixture: 'cpp-hazards.cpp', hazard: 'poll(' }, // run_forever: while(true) + { lang: 'C#', fixture: 'csharp-hazards.cs' }, // no non-terminating loop in the fixture + { lang: 'Java', fixture: 'java-hazards.java', hazard: 'ready(' }, // serve: while(true) + { lang: 'Go', fixture: 'go-hazards.go', hazard: 'handle(v)' }, // forInfinite: for{} +]; + +// ── Remaining-language worker-mode PDG (#2195 capstone) ───────────────────── +// +// The same both-sinks worker proof, run for the eight languages whose CFG +// visitors completed the PDG-language rollout AFTER the C-family: the dynamic +// languages (Python, PHP, Ruby), the systems/app languages (Rust, Swift, +// Kotlin, Dart), AND Vue (whose provider reuses the TypeScript CfgVisitor — the +// .vue file routes through the worker's Vue→TypeScript grammar mapping and the +// SFC +`; + const cfgs = cfgsOfSfc(sfc); + const loop = cfgs.find((c) => c.blocks.some((b) => b.text.includes('sum = sum + x'))); + expect(loop).toBeDefined(); + if (!loop) return; + + // The non-terminating loop has a back-edge but EXIT must still be reachable + // from EVERY block (the structural escape edge feeds the post-dom pass). + expect(edgeKinds(loop).has('loop-back')).toBe(true); + expect(isExitReachableFromAllBlocks(loop)).toBe(true); + + // Control dependence is computable and non-empty — the worker's CDG pass + // would emit > 0 edges for this function (matches the pipeline assertion). + const cd = computeControlDependence(loop); + expect(cd.edges.length).toBeGreaterThan(0); + for (const e of cd.edges) { + expect(['T', 'F']).toContain(e.label); + } + }); +}); From 5e96a99b0deb60bb3ea78f0006615a22be63f887 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 15 Jun 2026 14:29:21 +0100 Subject: [PATCH 02/26] feat(cfg): model value-position branches as control dependence (#2205, #2207) (#2211) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(cfg): model Java value-position switch as control flow (#2207) A value-position `switch` expression with ≥2 arms is now modeled as a CFG dispatch in the two highest-value carriers, instead of collapsing the owning statement to a single inline block: - `var x = switch (k) { … }` — the arms become real blocks reached by `switch-case` edges and rejoin at a binding continuation that carries the declared name's def (uses stay on the arm blocks). - `return switch (k) { … }` — each arm returns the function result, threading every active finalizer. This makes the arms control-dependent on the dispatch (the point of #2207 — they previously produced zero CDG), mirroring the Kotlin / Rust value-position binding pattern. `breaksBlock` routes a value-switch declaration out of `visitSeq` coalescing; `visitReturn` and `visitStmt` gain the carrier handling; `java-harvest` gains `bindingDefFacts`. An assignment RHS (`x = switch …`), a call argument, and a multi- declarator decl remain inline (documented gap). Java has no value- position `if` (the ternary is excluded, like Kotlin's elvis). Verified: 51 java-visitor tests (4 new), full CFG unit+integration suites (661) green, CDG snapshot byte-identical, bench --check PASS. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): model C# value-position switch expression as control flow (#2207) A C# `switch_expression` (`k switch { p => v, … }`) with ≥2 arms is now modeled as a CFG `switch-case` dispatch — a discriminant block, each arm value a block reached by a dispatch edge, all arms rejoining at one exit — in the three value-position carriers, instead of collapsing the owning construct to a single inline block: - `var x = k switch { … }` — arms rejoin at a binding continuation that carries the declared name's def (discriminant + arm uses on the arms). - `return k switch { … }` — each arm returns the function result, threading every active finalizer. - `=> k switch { … }` expression-bodied member — each arm returns. The arms are now control-dependent on the discriminant (the point of #2207 — they previously produced zero CDG). Arm patterns / `when` guards are harvested as conditional uses on the dispatch; an unguarded `_`/`var` arm is the exhaustive catch-all (a non-exhaustive switch keeps EXIT reachable via a no-match edge). `switch_expression` is distinct from `switch_statement`, so this adds a dedicated `visitSwitchExpr`. An assignment RHS (`x = k switch …`), a call argument, and a multi- declarator decl remain inline (documented gap). `csharp-harvest` gains `bindingDefFacts`. Verified: 47 csharp-visitor tests (4 new), full CFG unit+integration suites (664) green, CDG snapshot byte-identical, bench --check PASS. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): model PHP value-position match expression as control flow (#2207) A PHP `match($v) { c => v, default => v }` with ≥2 arms is now modeled as a CFG `switch-case` dispatch — a discriminant block, each arm value a block reached by a dispatch edge, all arms rejoining at one exit (no fallthrough) — in the two value-position carriers, instead of collapsing the owning statement to one inline block: - `$x = match($v) { … }` — the dominant PHP idiom (no typed local decl): arms rejoin at a binding continuation carrying the assignment target's def (condition + arm uses on the arms). - `return match($v) { … }` — each arm returns the function result, threading every active finally. The arms are now control-dependent on the discriminant (the point of #2207). Arm `match_condition_list`s are harvested as conditional uses on the dispatch; a `default` arm is the catch-all (a defaultless `match` throws UnhandledMatchError, kept EXIT-reachable via a no-match edge). `php-harvest` gains `assignmentDefFacts`. A `match` in a call argument / nested subexpression stays inline; the ternary `?:` is excluded by design (a micro-branch, like elvis). Verified: 37 php-visitor tests (3 new), CDG snapshot byte-identical, bench --check PASS. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): model Dart value-position switch expression as control flow (#2207) A Dart 3 value-position `switch (v) { p => e, _ => e }` with ≥2 arms is now modeled as a CFG `switch-case` dispatch — a discriminant block, each arm value a block reached by a dispatch edge, all arms rejoining at one exit (no fallthrough) — in the two value-position carriers, instead of collapsing the owning statement to one inline block: - `var x = switch (v) { … }` — single-binding decl; arms rejoin at a binding continuation carrying the declared name's def. - `return switch (v) { … }` — each arm returns the function result, threading every active finalizer. The arms are now control-dependent on the discriminant (the point of #2207). A Dart call value parses as `identifier` + `selector` (multiple children, not one node), so the arm-value facts come from a dedicated `switchExprArmValueFacts`; arm patterns harvest conditionally onto the dispatch; a `_` arm is the catch-all (a non-exhaustive switch keeps EXIT reachable via a no-match edge). `dart-harvest` gains `bindingDefFacts` + the arm value/pattern fact helpers. A `switch_expression` in a call argument / multi-binding decl stays inline (its conditional arm sub-evaluation remains the #2206 harvest may-def path, re-pointed in the regression test). `?:`/`??`/`?.` excluded by design. Verified: 39 dart-visitor tests (4 new), CDG snapshot byte-identical, bench --check PASS. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): model Swift value-position if/switch as control flow (#2207) A Swift 5.9 value-position `if`/`switch` is now modeled as control flow in the two value-position carriers, instead of collapsing the owning statement to one inline block: - `let x = if … else … / switch v { … }` — arms rejoin at a binding continuation carrying the declared name's def (condition + arm uses on the branch blocks). - `return if … / switch …` — each arm returns the function result, threading every active finalizer. The arms are now control-dependent on the branch (the point of #2207). tree-sitter-swift reuses `if_statement` / `switch_statement` for the value form (no separate `if_expression`/`switch_expression`), so the existing `visitIf`/`visitSwitch` are reused — this mirrors the Kotlin carrier exactly. `swift-harvest` gains `bindingDefFacts`. A value-position `if` requires an `else`; a value `switch` needs ≥2 entries. A value branch in a call argument / interpolation stays inline; `?:`/`??` are excluded by design. Verified: 31 swift-visitor tests (4 new), CDG snapshot byte-identical, bench --check PASS. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(cfg): model Kotlin assignment-RHS and value-position try as control flow (#2205) Completes the value-position branch carriers #2205 left deferred after the initial val/var-binding + return + expr-body work: - `x = when (k) { … }` / `x = if (c) a else b` / `x = try { … }` — a plain `=` assignment whose RHS is a modelable branch now models the arms as control flow and binds the LHS target at the rejoin (a compound `+=` and a plain-call RHS stay inline). - `val x = try { … } catch { … }` — a value-position `try` is now a modelable value branch (reusing visitTry), so the binding/assignment carriers route it through control flow too. The arms are now control-dependent on the branch (the point of #2205). `isModelableValueBranch` gains `try_expression`; `visitBranchExpr` routes it to `visitTry`; `isControlFlow`/`visitStmt` gain the `assignment` carrier; `kotlin-harvest` gains `assignmentDefFacts`. A branch nested in a call argument (`f(when …)`) stays inline (the direct value is the call); `?:`/`?.` micro-branches excluded by design. The `return try { … }` carrier is intentionally left out (finalizer-threading in return position is risky and was not requested). Verified: 43 kotlin-visitor tests (5 new), full CFG unit+integration suites (676) green, CDG snapshot byte-identical, bench --check PASS. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(cfg): Java colon-form value switch — yield ends the arm, no fallthrough (#2211) Tri-review (adversarial + correctness lanes) found that a value-position colon-form switch expression — `int x = switch(k){ case 1: yield a(); case 2: yield b(); }` (valid Java 14+) — reused the statement `visitSwitch` fallthrough logic, wiring a spurious `fallthrough` edge between the yield-terminated colon groups. A switch EXPRESSION never falls through between arms; the false edge dropped an arm's control-dependence edge and added a false reaching-defs propagation edge (verified by a real-parser probe). Arrow-form value switches were already correct. Root cause: `visitYield` modeled `yield e;` as a block that CONTINUES to the next statement. Semantically `yield` produces the switch-expression's value and EXITS the switch. Fix: `visitYield` now terminates the arm, jumping to the enclosing switch's exit and threading any finalizer it crosses — exactly like a `break` out of the switch, but carrying the yielded value's facts. Adds `ControlFlowContext.resolveYield()` (nearest SWITCH frame, never an intervening loop). `yield` is Java-only here (C# `yield return` is iterator semantics, untouched). Tests: a colon-form value switch asserting NO `fallthrough` edge and that BOTH arms are control-dependent on the dispatch (specific controller→ dependent pairs), plus a `return switch(…)` inside `try/finally` asserting `finally-return` threading per arm. Verified: full CFG unit+integration suites green, CDG snapshot byte- identical, bench --check PASS. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(cfg): Dart value switch — guarded `_` is not a catch-all; guard is a dispatch test (#2211) Tri-review (adversarial + correctness lanes) found two issues in the Dart `visitSwitchExpr` value-position modeling: 1. Catch-all detection was `pattern.text === '_'`, ignoring guards. A guarded `_ when c => …` is NOT exhaustive (Dart throws at runtime if no arm + guard matches), so falsely treating it as a catch-all suppressed the conservative no-match edge — asserting an exhaustive switch that isn't. The sibling C# visitor already gated catch-all on `!guard`. 2. A `when` guard parses as a bare sibling between the pattern and the value (no wrapper node), so it fell into the arm-VALUE children and was harvested as an unconditional arm-value use instead of a conditional dispatch test. Fix: new `armParts()` splits a `switch_expression_case` at the `=>` token into pattern / guard(s) / value(s). The pattern AND guard are harvested conditionally onto the dispatch block (they evaluate before the body, only when earlier arms missed); the arm-value facts come from the post-`=>` children only; the catch-all is gated on an unguarded `_`. Removes the now- unused dart-harvest `switchExprArm{Value,Pattern}Facts` (the visitor harvests per-child via the existing `facts`/`factsConditional`). Tests: a guarded value switch asserting the no-match edge is present (3 switch-case successors from the dispatch) and EXIT stays reachable, plus a test that the guard's use is recorded on the dispatch block, not an arm. Verified: full CFG suites green, CDG snapshot byte-identical, bench --check PASS. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(cfg): Kotlin return try {…} models the value-position try (#2205, #2211) Tri-review (maintainability + testing lanes) caught a doc-vs-code mismatch: the visitor header documents a `return ` carrier that includes `try`, but `visitReturn` only matched `when_expression`/`if_expression`, so `return try { … } catch { … }` fell through to the single inline-block path — its arms were not modeled. Fix: `visitReturn` now also matches `try_expression`, making the `return` carrier uniform with the binding / assignment / expression-body carriers (all route a value-position `try` through `isModelableValueBranch` → `visitTry`, threading the active finalizers per arm). No new machinery — just completes the carrier set the docstring already claimed. Tests: `return try {…} catch {…}` (throw + return edges, CDG-bearing, EXIT reachable) and `x = try {…} catch {…}` assignment-RHS (the assignment carrier's try path, previously untested). Verified: full CFG suites green, CDG snapshot byte-identical, bench --check PASS. Co-Authored-By: Claude Opus 4.8 (1M context) * test(cfg): cover the no-match edge of non-exhaustive C#/PHP value switches (#2211) The testing lane noted the `hasCatchAll`/`hasDefault === false` branch — where a value-position `switch`/`match` with no catch-all/default arm adds a conservative no-match edge so EXIT stays reachable — was untested (every existing fixture used a `_`/`default` arm). Adds a C# `x switch { 1 => …, 2 => … }` and a PHP `match($x){ 1 => …, 2 => … }` (no default) test, each asserting the dispatch fans to (arms + 1) `switch-case` successors and `isExitReachableFromAllBlocks` holds. Verified: full CFG suites green. Co-Authored-By: Claude Opus 4.8 (1M context) * docs(cfg): clarify Kotlin value-position try gate covers the finally-only path (#2211) Tri-review (maintainability lane) noted the inline comment at the `try_expression` branch of `isModelableValueBranch` said only "a catch's value", but the gate fires on `catch_block || finally_block`. Reword to acknowledge that a value-position `try` with a `catch` OR a `finally` is a modelable branch. Comment-only; no behavior change. Co-Authored-By: Claude Opus 4.8 (1M context) * docs(cfg): document the deliberate C# declaratorInit duplication (#2211) Tri-review (maintainability lane) flagged the byte-identical `declaratorInit` helper in `csharp.ts` (visitor) and `csharp-harvest.ts` (harvester). The two are standalone classes with no shared base (repo convention) and the only module both import is the generic `utils/ast-helpers` (types only) — not a home for a C#-grammar-specific helper. Resolve the lowest-risk way: a cross-reference comment at each definition noting the deliberate duplication and the keep-in-sync requirement. No new shared module; no behavior change. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(cfg): unwrap the paren in Dart visitSwitch for dispatch consistency (#2211) Tri-review (maintainability lane) noted `visitSwitch` (statement form) used the raw parenthesized `condition` (dispatch text `switch (x)`) while the new `visitSwitchExpr` unwraps it (`switch x`). Probed the vendored tree-sitter-dart: the `switch_statement` condition IS a `parenthesized_expression`, so apply `unwrapParen` in `visitSwitch` too. The harvest walks into the paren either way, so the discriminant's def/use facts are unchanged — only the dispatch block's text string normalizes. Verified byte-identical (cdg-snapshot + bench --check unchanged; the Dart unit tests assert topology, not block text). Co-Authored-By: Claude Opus 4.8 (1M context) * test(cfg): Swift single-entry value switch stays inline (#2211) Tri-review (testing lane) noted the existing Swift "stays inline" test used a plain call (`let x = g()`), which never exercises the single-entry switch gate. Add a real one-entry value switch (`let x = switch v { default: g() }`) asserting it coalesces (no switch-case edge), pinning the `>= 2` switch_entry threshold in `isModelableValueBranch`. Co-Authored-By: Claude Opus 4.8 (1M context) * test(cfg): cover the Kotlin expression-body try carrier (#2205, #2211) Tri-review (testing lane) noted the `fun f() = try { … } catch { … }` expression-body carrier (visitExprBody -> isModelableValueBranch accepting try_expression) existed but was untested. Add a regression asserting the expr-body try is modeled (throw + return edges, CDG-bearing, EXIT reachable). Co-Authored-By: Claude Opus 4.8 (1M context) * test(cfg): pin the value-branch carriers never throw on a truncated AST (R4) (#2211) Tri-review (adversarial lane) noted the new value-position branch carriers must return undefined / never throw on a malformed AST, or a single bad function would drop the whole file's CFG group (the R4 invariant). Add a per-language regression feeding a TRUNCATED value-branch carrier (an unterminated `var x = switch/match/if/when (…)`) through the existing `collectFunctions` + `buildFunctionCfg(...).not.toThrow()` graceful-undefined harness, for all six languages whose value-branch path is new (Java/C#/PHP/Dart/Swift/Kotlin). Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: Claude Opus 4.8 (1M context) --- .../ingestion/cfg/control-flow-context.ts | 12 ++ .../ingestion/cfg/visitors/csharp-harvest.ts | 29 ++- .../src/core/ingestion/cfg/visitors/csharp.ts | 188 +++++++++++++++++- .../ingestion/cfg/visitors/dart-harvest.ts | 23 +++ .../src/core/ingestion/cfg/visitors/dart.ts | 180 ++++++++++++++++- .../ingestion/cfg/visitors/java-harvest.ts | 18 ++ .../src/core/ingestion/cfg/visitors/java.ts | 136 +++++++++++-- .../ingestion/cfg/visitors/kotlin-harvest.ts | 19 ++ .../src/core/ingestion/cfg/visitors/kotlin.ts | 86 ++++++-- .../ingestion/cfg/visitors/php-harvest.ts | 18 ++ .../src/core/ingestion/cfg/visitors/php.ts | 156 ++++++++++++++- .../ingestion/cfg/visitors/swift-harvest.ts | 18 ++ .../src/core/ingestion/cfg/visitors/swift.ts | 85 ++++++++ gitnexus/test/unit/cfg/csharp-visitor.test.ts | 73 ++++++- gitnexus/test/unit/cfg/dart-visitor.test.ts | 88 +++++++- gitnexus/test/unit/cfg/java-visitor.test.ts | 103 +++++++++- gitnexus/test/unit/cfg/kotlin-visitor.test.ts | 79 ++++++++ gitnexus/test/unit/cfg/php-visitor.test.ts | 55 ++++- gitnexus/test/unit/cfg/swift-visitor.test.ts | 72 +++++++ 19 files changed, 1370 insertions(+), 68 deletions(-) diff --git a/gitnexus/src/core/ingestion/cfg/control-flow-context.ts b/gitnexus/src/core/ingestion/cfg/control-flow-context.ts index 6b9bf6c1c..0bfdef425 100644 --- a/gitnexus/src/core/ingestion/cfg/control-flow-context.ts +++ b/gitnexus/src/core/ingestion/cfg/control-flow-context.ts @@ -126,6 +126,18 @@ export class ControlFlowContext { ); } + /** + * Resolve a Java `yield e` (switch-EXPRESSION arm exit): the nearest enclosing + * SWITCH frame's exit, threading the finalizers stacked above it. Unlike a + * `break`, a `yield` ALWAYS targets the switch — never an intervening loop — so + * it cannot match a loop frame (a `yield` inside a loop inside a switch arm + * still exits the whole switch). Returns `undefined` when there is no enclosing + * switch (malformed input); the caller falls back to its conservative routing. + */ + resolveYield(): JumpResolution | undefined { + return this.resolve((f) => f.kind === 'switch'); + } + /** Every active finalizer, innermost first — what a `return` must cross. */ finalizersForReturn(): readonly FinalizerFrame[] { const fins: FinalizerFrame[] = []; diff --git a/gitnexus/src/core/ingestion/cfg/visitors/csharp-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/csharp-harvest.ts index 5e0296bf3..bcc615ab2 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/csharp-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/csharp-harvest.ts @@ -220,6 +220,27 @@ export class CsharpHarvester extends ScopeTreeHarvester { return acc.finish(); } + /** + * Def-ONLY facts for a value-position binding carrier (`var x = k switch {…}`, + * #2207): just the declared name(s)' def, attached to the continuation block the + * switch arms rejoin. The discriminant + arm-value USES are already harvested + * onto the branch's own blocks ({@link facts} on each arm), so this must NOT + * re-walk the initializer — only each `variable_declarator`'s name is a def here. + */ + bindingDefFacts(stmt: SyntaxNode): StatementFacts | undefined { + const acc = new FactAccumulator(stmt.startPosition.row + 1); + const decl = stmt.namedChildren.find((c) => c.type === 'variable_declaration'); + if (decl) { + for (let i = 0; i < decl.namedChildCount; i++) { + const d = decl.namedChild(i); + if (d?.type !== 'variable_declarator') continue; + const name = d.childForFieldName('name'); + if (name) this.def(name, acc); + } + } + return acc.defCount() ? acc.finish() : undefined; + } + /** Facts for a `foreach (decl in right)` head: decl binds, right is used. */ forEachHeadFacts(stmt: SyntaxNode): StatementFacts { const acc = new FactAccumulator(stmt.startPosition.row + 1); @@ -548,7 +569,13 @@ export class CsharpHarvester extends ScopeTreeHarvester { return { path, rootIdx }; } - /** The initializer value of a `variable_declarator` — the named child after `name`. */ + /** + * The initializer value of a `variable_declarator` — the named child after + * `name`. NOTE: deliberately duplicated in `csharp.ts` (the visitor is a + * standalone class with no shared base — repo convention). The two copies must + * stay in sync; there is no C#-specific shared module to host it, and the only + * module both files share is the generic `utils/ast-helpers` (types only). + */ private declaratorInit(declarator: SyntaxNode): SyntaxNode | undefined { const name = declarator.childForFieldName('name'); for (let i = 0; i < declarator.namedChildCount; i++) { diff --git a/gitnexus/src/core/ingestion/cfg/visitors/csharp.ts b/gitnexus/src/core/ingestion/cfg/visitors/csharp.ts index 49b6a6dbc..c3da50fbc 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/csharp.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/csharp.ts @@ -66,6 +66,12 @@ * unresolved label. * - Async/await suspension points are modeled as straight-line (the awaited * continuation is not a separate flow), consistent with the TS visitor. + * - A value-position `switch_expression` (`k switch {…}`) with ≥2 arms IS modeled + * as a `switch-case` dispatch in three carriers (#2207): a single-declarator + * `var x = k switch {…}` (arms rejoin at a binding continuation), `return k + * switch {…}`, and an `=> k switch {…}` expression body (each arm returns). + * A value switch in any OTHER position — an assignment RHS (`x = k switch …`), + * a call argument, or a multi-declarator decl — stays INLINE (one block). * - Def/use harvest scope: see `csharp-harvest.ts` — member/element writes are * not scalar defs; nested-function bodies are opaque in both directions. * @@ -186,7 +192,7 @@ class CsharpCfgWalk { dangling = [...scope.exits]; break; // the rest of the sequence is consumed by the dispose scope } - if (CONTROL_FLOW_TYPES.has(stmt.type)) { + if (this.breaksBlock(stmt)) { openSimple = undefined; // close any open straight-line block const res = this.visitStmt(stmt); if (res === null) continue; // transparent (empty nested block) @@ -222,9 +228,28 @@ class CsharpCfgWalk { }); } + /** + * Whether a statement breaks the current straight-line block. Adds the + * value-position switch carrier to the base {@link CONTROL_FLOW_TYPES} set: a + * `local_declaration_statement` whose single initializer is a modelable + * `switch_expression` (`var x = k switch {…}`, #2207) breaks so `visitStmt` + * models the arms as control flow instead of collapsing the decl to one block. + */ + private breaksBlock(stmt: SyntaxNode): boolean { + if (this.isValueSwitchDecl(stmt)) return true; + return CONTROL_FLOW_TYPES.has(stmt.type); + } + /** Dispatch one statement to its handler. Non-null except for empty blocks. */ visitStmt(stmt: SyntaxNode): SeqResult { switch (stmt.type) { + case 'local_declaration_statement': { + // `var x = k switch { … }` (#2207): the initializer is a value-position + // branch — model it as control flow and bind the result on the rejoin. + const branch = this.declValueSwitch(stmt); + if (branch) return this.visitBindBranch(stmt, branch); + return this.visitSimple(stmt); + } case 'if_statement': return this.visitIf(stmt); case 'while_statement': @@ -276,6 +301,18 @@ class CsharpCfgWalk { } private visitReturn(stmt: SyntaxNode): TraversalResult { + // `return k switch { … };` (#2207): the returned value is a value-position + // branch — model it as control flow, with each arm returning (its value IS + // the function result), threading every active finalizer per arm. + const branch = stmt.namedChildren.find((c) => c.type !== 'comment'); + if (branch && this.isModelableValueBranch(branch)) { + const res = this.visitBranchExpr(branch); + const finalizers = this.cfc.finalizersForReturn(); + for (const ex of res.exits) { + wireJumpThroughFinalizers(this.builder, ex, finalizers, this.builder.exitIndex, 'return'); + } + return { entry: res.entry, exits: [] }; + } const idx = this.builder.newBlock( startLineOf(stmt), endLineOf(stmt), @@ -656,6 +693,147 @@ class CsharpCfgWalk { return node.type.endsWith('_statement') || node.type === 'block'; } + // ── value-position switch expression (#2207) ──────────────────────────────── + + /** + * The `switch_expression` initializer of a single-declarator + * `local_declaration_statement` (`var x = k switch {…}`) when it is a modelable + * value branch, else undefined. A `using` decl and a multi-declarator decl are + * excluded (the `using` dispose path / multi-declarator stay inline). + */ + private declValueSwitch(stmt: SyntaxNode): SyntaxNode | undefined { + if (stmt.type !== 'local_declaration_statement') return undefined; + if (this.isUsingLocalDecl(stmt)) return undefined; + const decl = stmt.namedChildren.find((c) => c.type === 'variable_declaration'); + if (!decl) return undefined; + const declarators = decl.namedChildren.filter((c) => c.type === 'variable_declarator'); + if (declarators.length !== 1) return undefined; + const init = this.declaratorInit(declarators[0]); + return init && this.isModelableValueBranch(init) ? init : undefined; + } + + private isValueSwitchDecl(stmt: SyntaxNode): boolean { + return this.declValueSwitch(stmt) !== undefined; + } + + /** + * The initializer of a `variable_declarator` — its named child after `name`. + * NOTE: deliberately duplicated in `csharp-harvest.ts` (the harvester is a + * standalone class with no shared base — repo convention). The two copies must + * stay in sync; there is no C#-specific shared module to host it, and the only + * module both files share is the generic `utils/ast-helpers` (types only). + */ + private declaratorInit(declarator: SyntaxNode): SyntaxNode | undefined { + const name = declarator.childForFieldName('name'); + for (let i = 0; i < declarator.namedChildCount; i++) { + const c = declarator.namedChild(i); + if (c && c.id !== name?.id) return c; + } + return undefined; + } + + /** + * Whether `node` is a value-position branch worth modeling as control flow + * (#2207): a `switch_expression` (`k switch {…}`) with ≥2 arms — a real + * dispatch. C# value-position `if` does not exist (the ternary `?:` is excluded, + * like elvis in Kotlin). + */ + private isModelableValueBranch(node: SyntaxNode): boolean { + if (node.type !== 'switch_expression') return false; + return node.namedChildren.filter((c) => c.type === 'switch_expression_arm').length >= 2; + } + + /** + * Model a value-position `switch_expression` (`k switch { p => v, … }`) as a CFG + * dispatch: a discriminant block, each arm's value expression a block reached by + * a `switch-case` edge, all arms rejoining at a single exit. The arm patterns / + * `when` guards are harvested as conditional uses on the dispatch (a later arm + * test runs only when earlier arms didn't match), mirroring {@link visitSwitch}. + */ + private visitSwitchExpr(node: SyntaxNode): TraversalResult { + const arms = node.namedChildren.filter((c) => c.type === 'switch_expression_arm'); + const discriminant = node.namedChildren.find((c) => c.type !== 'switch_expression_arm') ?? node; + const dispatch = this.builder.newBlock( + startLineOf(node), + endLineOf(discriminant), + discriminant.text, + 'normal', + this.harvest.facts(discriminant), + ); + const switchExit = this.builder.newBlock(endLineOf(node), endLineOf(node), ''); + + let hasCatchAll = false; + for (const arm of arms) { + const pattern = arm.namedChild(0); + const guard = arm.namedChildren.find((c) => c.type === 'when_clause'); + if (pattern) this.builder.attachFacts(dispatch, this.harvest.factsConditional(pattern)); + if (guard) { + const inner = guard.namedChild(0); + if (inner) this.builder.attachFacts(dispatch, this.harvest.factsConditional(inner)); + } + // An unguarded `_`/`var` arm matches everything — the exhaustive default. + if (!guard && pattern && (pattern.type === 'discard' || pattern.type === 'var_pattern')) { + hasCatchAll = true; + } + const value = this.armValue(arm); + const armBlock = this.builder.newBlock( + startLineOf(value ?? arm), + endLineOf(value ?? arm), + (value ?? arm).text, + 'normal', + value ? this.harvest.facts(value) : undefined, + ); + this.builder.edge(dispatch, armBlock, 'switch-case'); + this.builder.edge(armBlock, switchExit, 'seq'); + } + // A non-exhaustive switch throws at runtime; conservatively keep EXIT directly + // reachable from the dispatch when no catch-all arm covers the no-match path. + if (!hasCatchAll) this.builder.edge(dispatch, switchExit, 'switch-case'); + + return { entry: dispatch, exits: [switchExit] }; + } + + /** The value expression of a `switch_expression_arm` (the child after `=>`). */ + private armValue(arm: SyntaxNode): SyntaxNode | undefined { + // pattern [when_clause] => value — the value is the LAST named child. + return arm.namedChild(arm.namedChildCount - 1) ?? undefined; + } + + /** Model a value-position branch as control flow (only `switch_expression`). */ + private visitBranchExpr(node: SyntaxNode): TraversalResult { + return this.visitSwitchExpr(node); + } + + /** + * An expression-bodied member's value (`=> k switch {…}`, #2207): if it is a + * modelable value branch, model its arms as control flow (each arm returns the + * function result); otherwise return null so the caller falls back to a single + * inline block. + */ + tryVisitValueBranchBody(expr: SyntaxNode): TraversalResult | null { + return this.isModelableValueBranch(expr) ? this.visitBranchExpr(expr) : null; + } + + /** + * `var x = k switch { … }` (#2207): visit the switch as control flow, then + * rejoin its arms at a facts-only continuation carrying ONLY the bound name's + * def (the discriminant + arm-value uses are already on the switch's blocks). + * The arms are now control-dependent on the dispatch, and `x` is defined at the + * join — mirrors the Java / Kotlin / Rust value-position binding. + */ + private visitBindBranch(stmt: SyntaxNode, branch: SyntaxNode): TraversalResult { + const res = this.visitBranchExpr(branch); + const cont = this.builder.newBlock( + startLineOf(stmt), + startLineOf(stmt), + '', + 'normal', + this.harvest.bindingDefFacts(stmt), + ); + this.builder.connect(res.exits, cont, 'seq'); + return { entry: res.entry, exits: [cont] }; + } + private visitTry(stmt: SyntaxNode): SeqResult { const bodyNode = stmt.childForFieldName('body'); const catchClauses: SyntaxNode[] = []; @@ -924,6 +1102,14 @@ function buildFunctionCfg(fnNode: SyntaxNode, filePath: string): FunctionCfg | u // Expression-bodied member / single-expression lambda: one block whose // value is returned. For an arrow clause the value is its inner expression. const expr = body.type === 'arrow_expression_clause' ? (body.namedChild(0) ?? body) : body; + // `=> k switch { … }` (#2207): model the arms as control flow, each arm + // returning the function result, instead of one inline block. + const branchRes = new CsharpCfgWalk(builder, harvest).tryVisitValueBranchBody(expr); + if (branchRes) { + builder.edge(builder.entryIndex, branchRes.entry, 'seq'); + builder.connect(branchRes.exits, builder.exitIndex, 'return'); + return builder.finish(harvest.bindingTable()); + } const blk = builder.newBlock( startLineOf(expr), endLineOf(expr), diff --git a/gitnexus/src/core/ingestion/cfg/visitors/dart-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/dart-harvest.ts index ff26c6971..9234c9dd0 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/dart-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/dart-harvest.ts @@ -280,6 +280,29 @@ export class DartHarvester { return acc.finish(); } + /** + * Def-ONLY facts for a value-position binding carrier (`var x = switch (…) {…}`, + * #2207): just the declared name(s)' def, attached to the continuation block the + * switch arms rejoin. The subject + arm-value USES are already harvested onto + * the branch's own blocks, so this must NOT re-walk the value — only each + * `initialized_variable_definition`'s `name` (and trailing binders) is a def. + */ + bindingDefFacts(stmt: SyntaxNode): StatementFacts | undefined { + const acc = new FactAccumulator(stmt.startPosition.row + 1); + for (const def of stmt.namedChildren) { + if (def.type !== 'initialized_variable_definition') continue; + const name = def.childForFieldName('name'); + if (name) this.def(name, acc); + for (let i = 0; i < def.namedChildCount; i++) { + const c = def.namedChild(i); + if (c?.type !== 'initialized_identifier') continue; + const id = c.namedChildren.find((g) => g.type === 'identifier'); + if (id) this.def(id, acc); + } + } + return acc.defCount() ? acc.finish() : undefined; + } + /** * Facts for a `for` head. For-in: the loop var name is a def, the collection a * use. C-style: the init/condition/update sub-expressions are walked for diff --git a/gitnexus/src/core/ingestion/cfg/visitors/dart.ts b/gitnexus/src/core/ingestion/cfg/visitors/dart.ts index 64253055c..e4ab651df 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/dart.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/dart.ts @@ -88,10 +88,13 @@ * - a closure (`function_expression`) is collected as its OWN function by * `isFunction`, so its body gets a standalone CFG; in the ENCLOSING function it * is an opaque straight-line value (its body is not followed inline). - * - `switch_expression` / `if`-as-expression / `?:` / `??` / `?.` used as a VALUE - * are left INLINE inside their owning statement's block — their conditional - * sub-evaluation is a HARVEST may-def concern (see dart-harvest.ts), not a CFG - * split (consistent with the TS `&&`/`??` treatment). + * - a value-position `switch_expression` (Dart 3) with ≥2 arms IS modeled as a + * `switch-case` dispatch in two carriers (#2207): a single-binding `var x = + * switch (v) {…}` (arms rejoin at a binding continuation) and `return switch + * (v) {…}` (each arm returns). A `switch_expression` in any OTHER position — a + * call argument, a multi-binding decl — stays INLINE (its conditional arm + * sub-evaluation is a HARVEST may-def concern, see dart-harvest.ts). `?:` / + * `??` / `?.` micro-branches are excluded by design (like the TS treatment). * * Known limitations: * - block-scope shadowing in the harvest is flattened to one function table (see @@ -249,6 +252,12 @@ class DartCfgWalk { private isControlFlow(stmt: SyntaxNode): boolean { if (this.isLabelError(stmt)) return true; // a stray label sibling — queue it if (isThrowStatement(stmt) || isRethrowStatement(stmt)) return true; + // `var x = switch (v) { … }` (#2207): a value-position switch breaks so + // `visitStmt` models the arms as control flow instead of coalescing. + if (stmt.type === 'local_variable_declaration') { + const v = this.directValue(stmt); + return v !== undefined && this.isModelableValueBranch(v); + } return CONTROL_FLOW_TYPES.has(stmt.type); } @@ -273,6 +282,13 @@ class DartCfgWalk { if (isThrowStatement(stmt)) return this.visitThrow(stmt); if (isRethrowStatement(stmt)) return this.visitRethrow(stmt); switch (stmt.type) { + case 'local_variable_declaration': { + // `var x = switch (v) { … }` (#2207): the value is a value-position + // branch — model it as control flow and bind the result on the rejoin. + const value = this.directValue(stmt); + if (value && this.isModelableValueBranch(value)) return this.visitBindBranch(stmt, value); + return this.visitSimple(stmt); + } case 'if_statement': return this.visitIf(stmt); case 'for_statement': @@ -322,6 +338,18 @@ class DartCfgWalk { /** `return [expr];` — threads through every active finalizer before EXIT. */ private visitReturn(stmt: SyntaxNode): TraversalResult { + // `return switch (v) { … };` (#2207): the returned value is a value-position + // branch — model it as control flow, with each arm returning (its value IS + // the function result), threading every active finalizer per arm. + const branch = stmt.namedChildren.find((c) => !isComment(c)); + if (branch && this.isModelableValueBranch(branch)) { + const res = this.visitBranchExpr(branch); + const finalizers = this.cfc.finalizersForReturn(); + for (const ex of res.exits) { + wireJumpThroughFinalizers(this.builder, ex, finalizers, this.builder.exitIndex, 'return'); + } + return { entry: res.entry, exits: [] }; + } const idx = this.builder.newBlock( startLineOf(stmt), endLineOf(stmt), @@ -579,7 +607,12 @@ class DartCfgWalk { */ private visitSwitch(stmt: SyntaxNode): TraversalResult { const labels = this.takeLabels(); - const value = stmt.childForFieldName('condition'); + // The `condition` field is a `parenthesized_expression` (verified) — unwrap it + // so the dispatch text/discriminant matches the value-position `visitSwitchExpr` + // form (`switch x`, not `switch (x)`). The harvest walks into the paren either + // way, so the def/use facts are unchanged — only the block text normalizes. + const condRaw = stmt.childForFieldName('condition'); + const value = condRaw ? this.unwrapParen(condRaw) : undefined; const dispatch = this.builder.newBlock( startLineOf(stmt), value ? endLineOf(value) : startLineOf(stmt), @@ -705,6 +738,143 @@ class DartCfgWalk { return id?.text || undefined; } + // ── value-position switch expression (#2207) ──────────────────────────────── + + /** + * The direct value of a `local_variable_declaration` with a SINGLE + * `initialized_variable_definition` (`var x = `): its `value` field. + * Returns undefined for a multi-binding decl (`var a = …, b = …`) — modeling + * those arm-by-arm is out of scope, so they coalesce inline. + */ + private directValue(stmt: SyntaxNode): SyntaxNode | undefined { + const defs = stmt.namedChildren.filter((c) => c.type === 'initialized_variable_definition'); + if (defs.length !== 1) return undefined; + return defs[0].childForFieldName('value') ?? undefined; + } + + /** + * Whether `node` is a value-position branch worth modeling as control flow + * (#2207): a `switch_expression` (Dart 3) with ≥2 arms — a real dispatch. Dart's + * value-position `if` does not exist; the ternary `?:` is excluded by design. + */ + private isModelableValueBranch(node: SyntaxNode): boolean { + if (node.type !== 'switch_expression') return false; + return node.namedChildren.filter((c) => c.type === 'switch_expression_case').length >= 2; + } + + /** Model a value-position branch as control flow (only `switch_expression`). */ + private visitBranchExpr(node: SyntaxNode): TraversalResult { + return this.visitSwitchExpr(node); + } + + /** + * Model a value-position `switch (v) { p [when g] => e, _ => e }` (Dart 3) as a + * CFG dispatch: a discriminant block, each arm's value a block reached by a + * `switch-case` edge, all arms rejoining at one exit (no fallthrough). The arm + * PATTERN and any `when` GUARD are harvested as conditional uses on the dispatch + * (they evaluate before the body, only when earlier arms missed); a Dart call + * value parses as `identifier` + `selector` (multiple children), so the arm-value + * facts come from each post-`=>` child. Only an UNGUARDED `_` arm is the + * exhaustive catch-all — a guarded `_ when …` is NOT (the no-match path still + * needs the conservative edge), mirroring the C# `visitSwitchExpr`. + */ + private visitSwitchExpr(node: SyntaxNode): TraversalResult { + const condRaw = node.childForFieldName('condition'); + const cond = condRaw ? this.unwrapParen(condRaw) : node; + const dispatch = this.builder.newBlock( + startLineOf(node), + endLineOf(cond), + `switch ${cond.text}`, + 'normal', + this.harvest.facts(cond), + ); + const switchExit = this.builder.newBlock(endLineOf(node), endLineOf(node), ''); + + const arms = node.namedChildren.filter((c) => c.type === 'switch_expression_case'); + let hasCatchAll = false; + for (const arm of arms) { + const { pattern, guards, values } = this.armParts(arm); + // The pattern + `when` guard are conditional dispatch tests, NOT arm-value + // uses — harvest them onto the dispatch (mirrors casePatterns for switch_statement). + if (pattern) this.builder.attachFacts(dispatch, this.harvest.factsConditional(pattern)); + for (const g of guards) this.builder.attachFacts(dispatch, this.harvest.factsConditional(g)); + if (pattern && pattern.text === '_' && guards.length === 0) hasCatchAll = true; + const first = values[0] ?? arm; + const last = values[values.length - 1] ?? arm; + const armBlock = this.builder.newBlock( + startLineOf(first), + endLineOf(last), + values.map((c) => c.text).join('') || arm.text, + 'normal', + undefined, + ); + for (const v of values) this.builder.attachFacts(armBlock, this.harvest.facts(v)); + this.builder.edge(dispatch, armBlock, 'switch-case'); + this.builder.edge(armBlock, switchExit, 'seq'); + } + // A non-exhaustive Dart switch expression throws at runtime; conservatively + // keep EXIT reachable via a no-match edge when no `_` catch-all arm exists. + if (!hasCatchAll) this.builder.edge(dispatch, switchExit, 'switch-case'); + + return { entry: dispatch, exits: [switchExit] }; + } + + /** + * Split a `switch_expression_case` at the `=>` token: the PATTERN (first named + * child before `=>`), any `when` GUARD (named children between the pattern and + * `=>` — tree-sitter-dart parses the guard as a bare sibling, not a wrapper), + * and the VALUE expression (named children after `=>` — a Dart call is split + * across `identifier` + `selector`, hence an array). + */ + private armParts(arm: SyntaxNode): { + pattern: SyntaxNode | undefined; + guards: SyntaxNode[]; + values: SyntaxNode[]; + } { + const before: SyntaxNode[] = []; + const values: SyntaxNode[] = []; + let seenArrow = false; + for (let i = 0; i < arm.childCount; i++) { + const c = arm.child(i); + if (!c) continue; + if (!c.isNamed) { + if (c.text === '=>') seenArrow = true; + continue; + } + if (isComment(c)) continue; + (seenArrow ? values : before).push(c); + } + return { pattern: before[0], guards: before.slice(1), values }; + } + + /** Strip a `parenthesized_expression` wrapper (a switch/if condition). */ + private unwrapParen(node: SyntaxNode): SyntaxNode { + if (node.type === 'parenthesized_expression') { + const inner = node.namedChildren.find((c) => !isComment(c)); + if (inner) return inner; + } + return node; + } + + /** + * `var x = switch (v) { … }` (#2207): visit the switch as control flow, then + * rejoin its arms at a facts-only continuation carrying ONLY the declared name's + * def (the subject + arm-value uses are already on the switch's blocks). The + * arms are now control-dependent on the dispatch — mirrors Java / Kotlin / Rust. + */ + private visitBindBranch(stmt: SyntaxNode, branch: SyntaxNode): TraversalResult { + const res = this.visitBranchExpr(branch); + const cont = this.builder.newBlock( + startLineOf(stmt), + startLineOf(stmt), + '', + 'normal', + this.harvest.bindingDefFacts(stmt), + ); + this.builder.connect(res.exits, cont, 'seq'); + return { entry: res.entry, exits: [cont] }; + } + // ── try / on / catch / finally ───────────────────────────────────────────── /** diff --git a/gitnexus/src/core/ingestion/cfg/visitors/java-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/java-harvest.ts index 549d212f7..c1e1c7e48 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/java-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/java-harvest.ts @@ -196,6 +196,24 @@ export class JavaHarvester extends ScopeTreeHarvester { return acc.finish(); } + /** + * Def-ONLY facts for a value-position binding carrier (`var x = switch (…) {…}`, + * #2207): just the declared name(s)' def, attached to the continuation block the + * switch arms rejoin. The switch subject + arm-value USES are already harvested + * onto the branch's own blocks ({@link facts} on each arm), so this must NOT + * re-walk the value — only each `variable_declarator`'s `name` is a def here. + */ + bindingDefFacts(stmt: SyntaxNode): StatementFacts | undefined { + const acc = new FactAccumulator(stmt.startPosition.row + 1); + for (let i = 0; i < stmt.namedChildCount; i++) { + const d = stmt.namedChild(i); + if (d?.type !== 'variable_declarator') continue; + const name = d.childForFieldName('name'); + if (name) this.def(name, acc); + } + return acc.defCount() ? acc.finish() : undefined; + } + /** Facts for a `for (T name : value)` head: name binds, value is used. */ forEachHeadFacts(stmt: SyntaxNode): StatementFacts { const acc = new FactAccumulator(stmt.startPosition.row + 1); diff --git a/gitnexus/src/core/ingestion/cfg/visitors/java.ts b/gitnexus/src/core/ingestion/cfg/visitors/java.ts index f9ba90a7a..236185d64 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/java.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/java.ts @@ -74,11 +74,12 @@ * TS `visitTry` over-approximation. * * Known limitations: - * - `switch` as an EXPRESSION value (`int r = switch (x) { … };`) is left INLINE - * inside its owning statement's block — its arms are not modeled as separate - * CFG blocks (the value flows to the assignment). Only a `switch` used as a - * STATEMENT (a direct statement child) is modeled as a dispatch construct. - * This mirrors the C# `switch_expression`-in-return handling — documented gap. + * - A value-position `switch` with ≥2 arms is modeled as control flow in the two + * highest-value carriers (#2207): a single-declarator `var x = switch (…) {…}` + * (arms rejoin at a binding continuation) and `return switch (…) {…}` (each arm + * returns). A value-position `switch` in any OTHER position — an assignment RHS + * (`x = switch …`), a call argument, or a multi-declarator decl — is still left + * INLINE inside its owning block (the value flows to one coalesced block). * - `yield` (in a switch expression) continues to the next statement (it yields * one value to the enclosing switch and the arm ends); the switch-expression * state machine is not modeled, consistent with the inline-value-switch gap. @@ -197,12 +198,7 @@ class JavaCfgWalk { let openSimple: number | undefined; for (const stmt of stmts) { - // A `switch_expression` only breaks a block when it is a STATEMENT switch. - // Used as a value (inside a declaration / return) it coalesces normally. - const breaks = - CONTROL_FLOW_TYPES.has(stmt.type) && - (stmt.type !== 'switch_expression' || this.isStatementSwitch(stmt)); - if (breaks) { + if (this.breaksBlock(stmt)) { openSimple = undefined; // close any open straight-line block const res = this.visitStmt(stmt); if (res === null) continue; // transparent (empty nested block) @@ -238,9 +234,35 @@ class JavaCfgWalk { }); } + /** + * Whether a statement breaks the current straight-line block. A + * `switch_expression` breaks only when it is a STATEMENT switch (a value- + * position switch used directly inside a `block` coalesces). A + * `local_variable_declaration` whose value is a modelable value-position switch + * (`var x = switch (…) {…}`, #2207) also breaks — `visitStmt` then models the + * arms as control flow instead of collapsing the decl to one inline block. + */ + private breaksBlock(stmt: SyntaxNode): boolean { + if (stmt.type === 'local_variable_declaration') { + const v = this.directValue(stmt); + return v !== undefined && this.isModelableValueBranch(v); + } + if (!CONTROL_FLOW_TYPES.has(stmt.type)) return false; + if (stmt.type === 'switch_expression') return this.isStatementSwitch(stmt); + return true; + } + /** Dispatch one statement to its handler. Non-null except for empty blocks. */ visitStmt(stmt: SyntaxNode): SeqResult { switch (stmt.type) { + case 'local_variable_declaration': { + // `var x = switch (k) { … }` (#2207): the value is a value-position + // branch — model it as control flow and bind the result on the rejoin, + // instead of collapsing the whole decl to one block. + const value = this.directValue(stmt); + if (value && this.isModelableValueBranch(value)) return this.visitBindBranch(stmt, value); + return this.visitSimple(stmt); + } case 'if_statement': return this.visitIf(stmt); case 'while_statement': @@ -289,6 +311,18 @@ class JavaCfgWalk { } private visitReturn(stmt: SyntaxNode): TraversalResult { + // `return switch (k) { … };` (#2207): the returned value is a value-position + // branch — model it as control flow, with each arm returning (its value IS + // the function result), threading every active finalizer per arm. + const branch = stmt.namedChildren.find((c) => !isComment(c)); + if (branch && this.isModelableValueBranch(branch)) { + const res = this.visitBranchExpr(branch); + const finalizers = this.cfc.finalizersForReturn(); + for (const ex of res.exits) { + wireJumpThroughFinalizers(this.builder, ex, finalizers, this.builder.exitIndex, 'return'); + } + return { entry: res.entry, exits: [] }; + } const idx = this.builder.newBlock( startLineOf(stmt), endLineOf(stmt), @@ -321,10 +355,13 @@ class JavaCfgWalk { } /** - * `yield e;` (switch-expression arm value) — yields one value to the enclosing - * switch and the arm ends; modeled as a block that continues to whatever - * follows (the switch-expression state machine is not modeled, see the visitor - * limitations). It carries the yielded value's def/use facts. + * `yield e;` (switch-expression arm value) — produces the switch-expression's + * value and EXITS the enclosing switch (it does NOT fall through to the next + * colon group). Modeled as a terminator that jumps to the switch exit, threading + * any finalizer it crosses — exactly like a `break` out of the switch but + * carrying the yielded value's def/use facts. (Reusing the statement `visitSwitch` + * for a value-position colon switch would otherwise wire a spurious `fallthrough` + * edge between yield-terminated arms — #2211 review.) */ private visitYield(stmt: SyntaxNode): TraversalResult { const idx = this.builder.newBlock( @@ -334,7 +371,13 @@ class JavaCfgWalk { 'normal', this.harvest.facts(stmt), ); - return { entry: idx, exits: [idx] }; + const res = this.cfc.resolveYield(); + const { target, finalizers } = res ?? { + target: this.builder.exitIndex, + finalizers: this.cfc.finalizersForReturn(), + }; + wireJumpThroughFinalizers(this.builder, idx, finalizers, target, 'break'); + return { entry: idx, exits: [] }; } private visitBreak(stmt: SyntaxNode): TraversalResult { @@ -717,6 +760,67 @@ class JavaCfgWalk { return label.namedChildren.filter((c) => !isComment(c)).length === 0; } + // ── value-position branches (#2207) ───────────────────────────────────────── + + /** + * The direct value of a `local_variable_declaration` with a SINGLE declarator: + * its `variable_declarator`'s `value` field (`var x = `). Returns + * undefined for a multi-declarator decl (`int a = …, b = …;`) — modeling those + * arm-by-arm is out of scope, so they coalesce inline. The DIRECT value only: + * `var x = f(switch …)` yields the call, not the nested switch, so an + * argument-position switch stays inline. + */ + private directValue(stmt: SyntaxNode): SyntaxNode | undefined { + const declarators = stmt.namedChildren.filter((c) => c.type === 'variable_declarator'); + if (declarators.length !== 1) return undefined; + return declarators[0].childForFieldName('value') ?? undefined; + } + + /** + * Whether `node` is a value-position branch worth modeling as control flow + * (#2207): a `switch_expression` with ≥2 case groups (a real dispatch). Java has + * no value-position `if` (the ternary `?:` is deliberately excluded, like elvis + * in Kotlin), so `switch` is the only carrier. + */ + private isModelableValueBranch(node: SyntaxNode): boolean { + if (node.type !== 'switch_expression') return false; + const body = node.childForFieldName('body'); + if (!body) return false; + const groups = body.namedChildren.filter( + (c) => c.type === 'switch_block_statement_group' || c.type === 'switch_rule', + ); + return groups.length >= 2; + } + + /** + * Model a value-position `switch` as control flow regardless of position — + * {@link visitSeq}'s `isStatementSwitch` gate keeps value-position switches + * inline, so call {@link visitSwitch} directly here. + */ + private visitBranchExpr(node: SyntaxNode): TraversalResult { + return this.visitSwitch(node); + } + + /** + * `var x = switch (k) { … }` (#2207): visit the switch as control flow, then + * rejoin its arms at a facts-only continuation carrying ONLY the bound name's + * def (the subject + arm-value uses are already harvested onto the switch's + * blocks). The arms are now control-dependent on the dispatch, and `x` is + * defined at the join — mirrors the Kotlin / Rust value-position binding. + */ + private visitBindBranch(stmt: SyntaxNode, branch: SyntaxNode): TraversalResult { + const res = this.visitBranchExpr(branch); + const cont = this.builder.newBlock( + startLineOf(stmt), + startLineOf(stmt), + '', + 'normal', + this.harvest.bindingDefFacts(stmt), + ); + this.builder.connect(res.exits, cont, 'seq'); + return { entry: res.entry, exits: [cont] }; + } + /** * try / catch / finally / try-with-resources. The `resources` of a * try-with-resources auto-close on BOTH normal and exception exit — exactly diff --git a/gitnexus/src/core/ingestion/cfg/visitors/kotlin-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/kotlin-harvest.ts index f226112f7..2d27b3d89 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/kotlin-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/kotlin-harvest.ts @@ -292,6 +292,25 @@ export class KotlinHarvester { return acc.defCount() ? acc.finish() : undefined; } + /** + * Def-ONLY facts for a value-position assignment carrier (`x = when (k) {…}`, + * #2205): just the LHS target, attached to the continuation block the branch + * arms rejoin. The branch subject + arm-value USES are already harvested onto + * the branch's own blocks, so this must NOT re-walk the RHS — only a plain `=` + * to a simple-identifier lvalue defines (a member / index target is not a + * scalar def; a compound `+=` is not a value-branch carrier). + */ + assignmentDefFacts(node: SyntaxNode): StatementFacts | undefined { + if (this.assignmentOperator(node) !== '=') return undefined; + const acc = new FactAccumulator(node.startPosition.row + 1); + const lvalue = node.namedChildren.find((c) => c.type === 'directly_assignable_expression'); + if (lvalue) { + const lv = this.unwrapAssignable(lvalue); + if (lv.type === 'simple_identifier') this.def(lv, acc); + } + return acc.defCount() ? acc.finish() : undefined; + } + /** ENTRY-block facts for the parameters (defs only). */ paramFacts(): StatementFacts | undefined { const acc = new FactAccumulator(this.fnNode.startPosition.row + 1); diff --git a/gitnexus/src/core/ingestion/cfg/visitors/kotlin.ts b/gitnexus/src/core/ingestion/cfg/visitors/kotlin.ts index d1889a260..68ef9f118 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/kotlin.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/kotlin.ts @@ -80,12 +80,14 @@ * Java/C#/TS over-approximation. * * Kotlin-specific modeling decisions (documented approximations): - * - `if` / `when` / `try` used as an EXPRESSION VALUE (assigned, returned inline, - * passed as an argument) is left INLINE inside its owning statement's block — - * its arms are not modeled as separate CFG blocks (the value flows to the - * consumer). Only a STATEMENT-position construct (a direct `statements` child) - * becomes a dispatch/branch construct. This mirrors the Java inline-value-switch - * gap — documented, not faked. + * - a value-position `if` (with `else`) / `when` (≥2 arms) / `try` IS modeled as + * control flow (#2205) in four carriers: a `val/var x = ` binding, an + * `x = ` assignment, a `return `, and a `fun f() = ` + * expression body — its arms become separate CFG blocks that rejoin at a + * binding/return continuation. A branch in any OTHER value position — nested in + * a call argument (`f(when …)`), a deeper subexpression — is left INLINE (the + * value flows to the consumer in one block). The ternary-like `?:` (elvis) and + * `?.` micro-branches are excluded by design. * - a `lambda_literal` / nested `anonymous_function` / nested * `function_declaration` is collected as its OWN function by `isFunction`, so * its body gets a standalone CFG; in the ENCLOSING function it is an opaque @@ -246,9 +248,10 @@ class KotlinCfgWalk { * Whether a statement breaks the current straight-line block. `if` / `when` / * `try` are EXPRESSIONS in Kotlin — they break a block when used as a STATEMENT * (a direct child of a `statements` list), OR when they are the value of a - * `val/var x = ` binding (#2205) — `visitStmt`'s `property_declaration` - * case then models the arms as control flow. Other value positions (an - * assignment RHS, a call argument) still coalesce — a remaining gap. + * `val/var x = ` binding or an `x = ` assignment (#2205) — + * `visitStmt`'s `property_declaration` / `assignment` case then models the arms + * as control flow. A call argument value position still coalesces (a remaining + * gap — the branch is nested in a call, harder to bind). */ private isControlFlow(stmt: SyntaxNode): boolean { if (stmt.type === 'label') return true; // queue label, emit no block @@ -256,6 +259,7 @@ class KotlinCfgWalk { const v = this.directValue(stmt); return v !== undefined && this.isModelableValueBranch(v); } + if (stmt.type === 'assignment') return this.assignmentBranch(stmt) !== undefined; if (!CONTROL_FLOW_TYPES.has(stmt.type)) return false; if (this.isExpressionConstruct(stmt.type)) return this.isStatementPosition(stmt); return true; @@ -309,6 +313,13 @@ class KotlinCfgWalk { if (value && this.isModelableValueBranch(value)) return this.visitBindBranch(stmt, value); return this.visitSimple(stmt); } + case 'assignment': { + // `x = when (k) { … }` / `x = if (c) a else b` / `x = try { … }` (#2205): + // model the RHS branch as control flow and bind the target on the rejoin. + const branch = this.assignmentBranch(stmt); + if (branch) return this.visitBindAssign(stmt, branch); + return this.visitSimple(stmt); + } default: return this.visitSimple(stmt); } @@ -345,11 +356,13 @@ class KotlinCfgWalk { /** `return [expr]` / `return@label` — threads through every active finalizer. */ private visitReturn(stmt: SyntaxNode): TraversalResult { - // `return when (k) { … }` / `return if (c) a else b` (#2205): the returned - // value is a value-position branch — model it as control flow, with each arm - // returning (its value IS the function result), threading finalizers per arm. + // `return when (k) { … }` / `return if (c) a else b` / `return try { … }` + // (#2205): the returned value is a value-position branch — model it as control + // flow, with each arm returning (its value IS the function result), threading + // finalizers per arm. const branch = stmt.namedChildren.find( - (c) => c.type === 'when_expression' || c.type === 'if_expression', + (c) => + c.type === 'when_expression' || c.type === 'if_expression' || c.type === 'try_expression', ); if (branch && this.isModelableValueBranch(branch)) { const res = this.visitBranchExpr(branch); @@ -470,16 +483,25 @@ class KotlinCfgWalk { return node.namedChildren.filter((c) => c.type === 'when_entry').length >= 2; } if (node.type === 'if_expression') return this.elseNodeOf(node) !== undefined; + // `val x = try { … } catch { … }` / `try { … } finally { … }` (#2205): a + // value-position `try` with a `catch` OR a `finally` is a real branch — its + // value is the body's value, a catch's value, or the body's value threaded + // through a finalizer — so model it as control flow. + if (node.type === 'try_expression') { + return node.namedChildren.some((c) => c.type === 'catch_block' || c.type === 'finally_block'); + } return false; } /** - * Model a value-position `when`/`if` as control flow regardless of its + * Model a value-position `when`/`if`/`try` as control flow regardless of its * statement/value position — {@link visitStmt}'s `isStatementPosition` gate keeps * value-position branches inline, so call the branch handlers directly here. */ private visitBranchExpr(node: SyntaxNode): TraversalResult { - return node.type === 'when_expression' ? this.visitWhen(node) : this.visitIf(node); + if (node.type === 'when_expression') return this.visitWhen(node); + if (node.type === 'try_expression') return this.visitTry(node) ?? this.visitSimple(node); + return this.visitIf(node); } /** @@ -502,6 +524,40 @@ class KotlinCfgWalk { return { entry: res.entry, exits: [cont] }; } + /** + * The value-position branch on a plain `=` assignment RHS (`x = when (k) {…}` / + * `x = if (c) a else b` / `x = try {…}`, #2205), or undefined. Only a plain `=` + * (not a compound `+=`) with a modelable-branch RHS qualifies. + */ + private assignmentBranch(stmt: SyntaxNode): SyntaxNode | undefined { + if (stmt.type !== 'assignment') return undefined; + const eq = stmt.children.find((c) => !c.isNamed && c.text === '='); + if (!eq) return undefined; // compound assignment (`+=` etc.) is not a carrier + const rhs = stmt.namedChildren.find( + (c) => c.type !== 'directly_assignable_expression' && !isComment(c), + ); + return rhs && this.isModelableValueBranch(rhs) ? rhs : undefined; + } + + /** + * `x = ` (#2205): visit the RHS branch as control flow, then rejoin its + * arms at a facts-only continuation carrying ONLY the LHS target def (the branch + * subject + arm-value uses are already on the branch's blocks). The arms are now + * control-dependent on the branch — mirrors the Ruby value-branch assignment. + */ + private visitBindAssign(stmt: SyntaxNode, branch: SyntaxNode): TraversalResult { + const res = this.visitBranchExpr(branch); + const cont = this.builder.newBlock( + startLineOf(stmt), + startLineOf(stmt), + '', + 'normal', + this.harvest.assignmentDefFacts(stmt), + ); + this.builder.connect(res.exits, cont, 'seq'); + return { entry: res.entry, exits: [cont] }; + } + /** * A `fun f() = EXPR` expression body (#2205). A value-position branch is modeled * as control flow (each arm yields the returned function result); any other diff --git a/gitnexus/src/core/ingestion/cfg/visitors/php-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/php-harvest.ts index 9bd680407..8e4062457 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/php-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/php-harvest.ts @@ -303,6 +303,24 @@ export class PhpHarvester { return acc.finish(); } + /** + * Def-ONLY facts for a value-position assignment carrier (`$x = match($v) {…}`, + * #2207): just the LHS target(s), attached to the continuation block the match + * arms rejoin. The match condition + arm-value USES are already harvested onto + * the branch's own blocks (visitMatch), so this must NOT re-walk the RHS. A + * member/subscript target (`$this->x = match …`) has no scalar def → undefined. + */ + assignmentDefFacts(assignExpr: SyntaxNode): StatementFacts | undefined { + const acc = new FactAccumulator(assignExpr.startPosition.row + 1); + const left = assignExpr.childForFieldName('left'); + if (left) { + const lv = this.unwrapParen(left); + if (lv.type === 'variable_name') this.def(lv, acc); + else if (lv.type === 'list_literal') for (const v of this.listTargets(lv)) this.def(v, acc); + } + return acc.defCount() ? acc.finish() : undefined; + } + /** Facts for a `foreach ($it as [$k =>] $v)` head: targets bind, iterable used. */ foreachHeadFacts(stmt: SyntaxNode): StatementFacts { const acc = new FactAccumulator(stmt.startPosition.row + 1); diff --git a/gitnexus/src/core/ingestion/cfg/visitors/php.ts b/gitnexus/src/core/ingestion/cfg/visitors/php.ts index f9a73c74f..3e4878513 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/php.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/php.ts @@ -44,8 +44,8 @@ * - loops (for / foreach / while / do-while) → `cond-true` / `loop-back` / * `cond-false` * - switch → `switch-case` / `fallthrough` (a `case` with no `break`/`return` - * falls through to the next case); `match` is left INLINE as a value - * (no fallthrough — see the limitations). + * falls through to the next case); a value-position `match` with ≥2 arms also + * dispatches as `switch-case` (no fallthrough), see the limitations. * - try/catch → `throw` (every protected-region block → the handler); a * `finally` runs on normal AND exception exit, so a `return`/`break`/`continue` * crossing it gets a `finally-*` completion edge. @@ -68,10 +68,12 @@ * region edges to the handler (an exception may fire mid-block). * * Known limitations: - * - `match` is a value-position EXPRESSION (`$r = match($x) { … }`), kept INLINE - * inside its owning statement's block — its arms are not modeled as separate - * CFG blocks (the value flows to the assignment). Documented gap, mirroring the - * Java inline-value-switch handling. + * - A value-position `match($x) { … }` with ≥2 arms IS modeled as a `switch-case` + * dispatch in two carriers (#2207): an `$x = match(…) {…}` assignment (arms + * rejoin at a binding continuation) and `return match(…) {…}` (each arm + * returns). A `match` in any OTHER position — a call argument, a nested + * subexpression — stays INLINE inside its owning block. The ternary `?:` is + * excluded by design (a micro-branch, like elvis in Kotlin). * - context-manager-style suppression and PHP's exception-from-mid-call outside * any `try` are not modeled (no edge), matching the other visitors. * - `goto` / named labels are modeled as straight-line blocks (the label is a @@ -192,7 +194,11 @@ class PhpCfgWalk { for (const stmt of stmts) { // An `expression_statement` wrapping a bare `throw_expression` is a // terminator (PHP has no `throw_statement` node), so it breaks the block. - const breaks = CONTROL_FLOW_TYPES.has(stmt.type) || this.isThrowStatement(stmt); + // An `$x = match($v) {…}` value-position assignment breaks too (#2207). + const breaks = + CONTROL_FLOW_TYPES.has(stmt.type) || + this.isThrowStatement(stmt) || + this.isValueBranchAssignment(stmt); if (breaks) { openSimple = undefined; // close any open straight-line block const res = this.visitStmt(stmt); @@ -232,6 +238,10 @@ class PhpCfgWalk { /** Dispatch one statement to its handler. Non-null except for empty blocks. */ visitStmt(stmt: SyntaxNode): SeqResult { if (this.isThrowStatement(stmt)) return this.visitThrow(stmt); + // `$x = match($v) { … };` (#2207): model the match arms as control flow and + // bind the assignment target on the rejoin. + const assign = this.assignmentBranch(stmt); + if (assign) return this.visitBindAssign(stmt, assign.expr, assign.match); switch (stmt.type) { case 'if_statement': return this.visitIf(stmt); @@ -285,6 +295,18 @@ class PhpCfgWalk { } private visitReturn(stmt: SyntaxNode): TraversalResult { + // `return match($v) { … };` (#2207): the returned value is a value-position + // branch — model it as control flow, with each arm returning (its value IS + // the function result), threading every active finally per arm. + const branch = stmt.namedChildren.find((c) => !isComment(c)); + if (branch && this.isModelableValueBranch(branch)) { + const res = this.visitBranchExpr(branch); + const finalizers = this.cfc.finalizersForReturn(); + for (const ex of res.exits) { + wireJumpThroughFinalizers(this.builder, ex, finalizers, this.builder.exitIndex, 'return'); + } + return { entry: res.entry, exits: [] }; + } const idx = this.builder.newBlock( startLineOf(stmt), endLineOf(stmt), @@ -670,6 +692,126 @@ class PhpCfgWalk { return group.childForFieldName('value') ?? undefined; } + // ── value-position match expression (#2207) ───────────────────────────────── + + /** + * The `{expr, match}` of an `$x = match($v) {…}` value-position assignment + * carrier, or undefined. `expr` is the `assignment_expression` (for the target + * def); `match` is the modelable `match_expression` RHS. Only a plain `=` + * assignment qualifies (an augmented `??=` etc. is not a value-branch bind). + */ + private assignmentBranch(stmt: SyntaxNode): { expr: SyntaxNode; match: SyntaxNode } | undefined { + if (stmt.type !== 'expression_statement') return undefined; + const expr = stmt.namedChildren.find((c) => !isComment(c)); + if (!expr || expr.type !== 'assignment_expression') return undefined; + const right = expr.childForFieldName('right'); + return right && this.isModelableValueBranch(right) ? { expr, match: right } : undefined; + } + + /** Whether a statement is an `$x = match(…) {…}` value-branch assignment. */ + private isValueBranchAssignment(stmt: SyntaxNode): boolean { + return this.assignmentBranch(stmt) !== undefined; + } + + /** + * Whether `node` is a value-position branch worth modeling as control flow + * (#2207): a `match_expression` with ≥2 arms — a real dispatch. PHP `match` is + * the only value-position branch (there is no `if`-expression); the ternary + * `?:` is deliberately excluded, like elvis in Kotlin. + */ + private isModelableValueBranch(node: SyntaxNode): boolean { + if (node.type !== 'match_expression') return false; + const block = node.childForFieldName('body'); + if (!block) return false; + return ( + block.namedChildren.filter( + (c) => c.type === 'match_conditional_expression' || c.type === 'match_default_expression', + ).length >= 2 + ); + } + + /** Model a value-position branch as control flow (only `match_expression`). */ + private visitBranchExpr(node: SyntaxNode): TraversalResult { + return this.visitMatch(node); + } + + /** + * Model a value-position `match($v) { c => v, default => v }` as a CFG dispatch: + * a discriminant block, each arm's value expression a block reached by a + * `switch-case` edge, all arms rejoining at one exit (no fallthrough — `match` + * never falls through). The arm condition lists are harvested as conditional + * uses on the dispatch (a later arm test runs only when earlier arms missed). + */ + private visitMatch(node: SyntaxNode): TraversalResult { + const condRaw = node.childForFieldName('condition'); + const cond = condRaw ? this.unwrapParen(condRaw) : node; + const dispatch = this.builder.newBlock( + startLineOf(node), + endLineOf(cond), + cond.text, + 'normal', + this.harvest.facts(cond), + ); + const matchExit = this.builder.newBlock(endLineOf(node), endLineOf(node), ''); + + const block = node.childForFieldName('body'); + const arms = block + ? block.namedChildren.filter( + (c) => c.type === 'match_conditional_expression' || c.type === 'match_default_expression', + ) + : []; + let hasDefault = false; + for (const arm of arms) { + const condList = arm.namedChildren.find((c) => c.type === 'match_condition_list'); + if (condList) this.builder.attachFacts(dispatch, this.harvest.factsConditional(condList)); + if (arm.type === 'match_default_expression') hasDefault = true; + const value = this.matchArmValue(arm); + const armBlock = this.builder.newBlock( + startLineOf(value ?? arm), + endLineOf(value ?? arm), + (value ?? arm).text, + 'normal', + value ? this.harvest.facts(value) : undefined, + ); + this.builder.edge(dispatch, armBlock, 'switch-case'); + this.builder.edge(armBlock, matchExit, 'seq'); + } + // `match` with no `default` throws `\UnhandledMatchError` on no match; keep + // EXIT reachable via a conservative no-match edge when no default arm exists. + if (!hasDefault) this.builder.edge(dispatch, matchExit, 'switch-case'); + + return { entry: dispatch, exits: [matchExit] }; + } + + /** The value (result) expression of a match arm — its LAST named child. */ + private matchArmValue(arm: SyntaxNode): SyntaxNode | undefined { + const named = arm.namedChildren.filter((c) => !isComment(c)); + return named[named.length - 1]; + } + + /** + * `$x = match($v) { … }` (#2207): visit the match as control flow, then rejoin + * its arms at a facts-only continuation carrying ONLY the LHS target def (the + * condition + arm-value uses are already on the match's blocks). The arms are + * now control-dependent on the dispatch — mirrors the Ruby value-branch assign. + */ + private visitBindAssign( + stmt: SyntaxNode, + assignExpr: SyntaxNode, + branch: SyntaxNode, + ): TraversalResult { + const res = this.visitBranchExpr(branch); + const cont = this.builder.newBlock( + startLineOf(stmt), + startLineOf(stmt), + '', + 'normal', + this.harvest.assignmentDefFacts(assignExpr), + ); + this.builder.connect(res.exits, cont, 'seq'); + return { entry: res.entry, exits: [cont] }; + } + /** * try / catch / finally. A `finally` runs on BOTH normal and exception exit — * a `return`/`break`/`continue` crossing it threads through it (`finally-*` diff --git a/gitnexus/src/core/ingestion/cfg/visitors/swift-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/swift-harvest.ts index 7e17a6eee..e9bba8bee 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/swift-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/swift-harvest.ts @@ -279,6 +279,24 @@ export class SwiftHarvester { return acc.finish(); } + /** + * Def-ONLY facts for a value-position binding carrier (`let x = if … / switch …`, + * #2207): just the declared name pattern's leaves, attached to the continuation + * block the branch arms rejoin. The condition + arm-value USES are already + * harvested onto the branch's own blocks (visitIf / visitSwitch), so this must + * NOT re-walk the value — only the `name`-field pattern leaves are defs here. + */ + bindingDefFacts(stmt: SyntaxNode): StatementFacts | undefined { + const acc = new FactAccumulator(stmt.startPosition.row + 1); + for (let i = 0; i < stmt.childCount; i++) { + if (stmt.fieldNameForChild(i) === 'name') { + const pat = stmt.child(i); + if (pat) this.defPattern(pat, acc); + } + } + return acc.defCount() ? acc.finish() : undefined; + } + /** * MAY-def facts for a `switch_pattern`'s value bindings (`case let n` / * `case .some(let v)`). The binding only takes effect when the case matches, diff --git a/gitnexus/src/core/ingestion/cfg/visitors/swift.ts b/gitnexus/src/core/ingestion/cfg/visitors/swift.ts index 57af723d4..050a48e39 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/swift.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/swift.ts @@ -88,6 +88,12 @@ * trailing closure, which is unwrapped to model scope-exit flow. * * Known limitations: + * - a value-position `if`/`switch` (Swift 5.9) IS modeled as control flow in two + * carriers (#2207): a `let x = if … else … / switch … {…}` binding (arms rejoin + * at a binding continuation) and `return if … / switch …` (each arm returns). + * tree-sitter-swift reuses `if_statement` / `switch_statement` for the value + * form. A value branch in any OTHER position (an argument, an interpolation) + * stays inline; the ternary `?:` / `??` are excluded by design. * - computed properties (`var y: Int { get { … } set { … } }`) have their bodies * inside `computed_getter` / `computed_setter` rather than a function node; v1 * does NOT build a CFG for them (documented gap, not faked). @@ -232,6 +238,12 @@ class SwiftCfgWalk { private isControlFlow(stmt: SyntaxNode): boolean { if (stmt.type === 'statement_label') return true; // queue label, emit no block if (this.isDeferCall(stmt)) return true; + // `let x = if … / switch …` (Swift 5.9, #2207): a value-position branch breaks + // so `visitStmt` models the arms as control flow instead of coalescing. + if (stmt.type === 'property_declaration') { + const v = this.directValue(stmt); + return v !== undefined && this.isModelableValueBranch(v); + } return CONTROL_FLOW_TYPES.has(stmt.type); } @@ -245,6 +257,13 @@ class SwiftCfgWalk { } if (this.isDeferCall(stmt)) return this.visitDefer(stmt); switch (stmt.type) { + case 'property_declaration': { + // `let x = if … / switch …` (Swift 5.9, #2207): the value is a value- + // position branch — model it as control flow and bind on the rejoin. + const value = this.directValue(stmt); + if (value && this.isModelableValueBranch(value)) return this.visitBindBranch(stmt, value); + return this.visitSimple(stmt); + } case 'if_statement': return this.visitIf(stmt); case 'guard_statement': @@ -308,6 +327,20 @@ class SwiftCfgWalk { /** `return [expr]` — threads through every active `defer` (LIFO) before EXIT. */ private visitReturn(stmt: SyntaxNode): TraversalResult { + // `return if … / switch …` (Swift 5.9, #2207): the returned value is a value- + // position branch — model it as control flow, with each arm returning (its + // value IS the function result), threading every active finalizer per arm. + const branch = stmt.namedChildren.find( + (c) => c.type === 'if_statement' || c.type === 'switch_statement', + ); + if (branch && this.isModelableValueBranch(branch)) { + const res = this.visitBranchExpr(branch); + const finalizers = this.cfc.finalizersForReturn(); + for (const ex of res.exits) { + wireJumpThroughFinalizers(this.builder, ex, finalizers, this.builder.exitIndex, 'return'); + } + return { entry: res.entry, exits: [] }; + } const idx = this.builder.newBlock( startLineOf(stmt), endLineOf(stmt), @@ -384,6 +417,58 @@ class SwiftCfgWalk { return labels; } + // ── value-position branches (#2207) ───────────────────────────────────────── + + /** + * The value-position branch of a `property_declaration` (`let x = if … / switch + * …`, Swift 5.9): the direct `if_statement` / `switch_statement` child (the value + * after `=`), or undefined. tree-sitter-swift reuses the statement nodes for the + * value form — there is no separate `if_expression` / `switch_expression`. + */ + private directValue(stmt: SyntaxNode): SyntaxNode | undefined { + return stmt.namedChildren.find( + (c) => c.type === 'if_statement' || c.type === 'switch_statement', + ); + } + + /** + * Whether `node` is a value-position branch worth modeling as control flow + * (#2207): an `if` with an `else` (a value-position `if` always has one), or a + * `switch` with ≥2 entries — a real dispatch. The ternary `?:` and `??` are + * excluded by design (micro-branches, like the Kotlin elvis). + */ + private isModelableValueBranch(node: SyntaxNode): boolean { + if (node.type === 'if_statement') return this.elseNodeOf(node) !== undefined; + if (node.type === 'switch_statement') { + return node.namedChildren.filter((c) => c.type === 'switch_entry').length >= 2; + } + return false; + } + + /** Model a value-position `if`/`switch` as control flow, bypassing position. */ + private visitBranchExpr(node: SyntaxNode): TraversalResult { + return node.type === 'switch_statement' ? this.visitSwitch(node) : this.visitIf(node); + } + + /** + * `let x = if … / switch …` (#2207): visit the branch as control flow, then + * rejoin its arms at a facts-only continuation carrying ONLY the bound name's + * def (the condition + arm-value uses are already on the branch's blocks). The + * arms are now control-dependent on the branch — mirrors Kotlin / Rust. + */ + private visitBindBranch(stmt: SyntaxNode, branch: SyntaxNode): TraversalResult { + const res = this.visitBranchExpr(branch); + const cont = this.builder.newBlock( + startLineOf(stmt), + startLineOf(stmt), + '', + 'normal', + this.harvest.bindingDefFacts(stmt), + ); + this.builder.connect(res.exits, cont, 'seq'); + return { entry: res.entry, exits: [cont] }; + } + // ── branches ────────────────────────────────────────────────────────────── /** diff --git a/gitnexus/test/unit/cfg/csharp-visitor.test.ts b/gitnexus/test/unit/cfg/csharp-visitor.test.ts index 01b7b058c..dabd1792f 100644 --- a/gitnexus/test/unit/cfg/csharp-visitor.test.ts +++ b/gitnexus/test/unit/cfg/csharp-visitor.test.ts @@ -196,14 +196,67 @@ describe('C# CfgVisitor — switch', () => { expect(edgeKinds(cfg).has('switch-case')).toBe(true); }); - it('switch_expression arms each dispatch as a guarded branch (switch-case)', () => { + it('return switch_expression: each arm dispatches and returns the result (#2207)', () => { const cfg = cs.cfgOf( `class C { int M(int x) { return x switch { 1 => a(), 2 => b(), _ => c() }; } }`, ); - // The switch-expression lives inside the return block — it does not break a - // basic block, but the function still has a well-formed single-exit CFG. - expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); expect(edgeKinds(cfg).has('return')).toBe(true); + // every arm reaches EXIT (its value IS the returned result). + expect(reaches(cfg, block(cfg, 'a()'), cfg.exitIndex)).toBe(true); + expect(reaches(cfg, block(cfg, 'c()'), cfg.exitIndex)).toBe(true); + // a() does NOT fall into b() (arms never fall through). + expect(reaches(cfg, block(cfg, 'a()'), block(cfg, 'b()'))).toBe(false); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('value-position switch declaration is modeled, def bound at the join (#2207)', () => { + const cfg = cs.cfgOf( + `class C { int M(int x) { var y = x switch { 1 => a(), _ => b() }; use(y); return 0; } }`, + ); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + // each arm rejoins and reaches the downstream use of the bound result. + expect(reaches(cfg, block(cfg, 'a()'), block(cfg, 'use(y);'))).toBe(true); + expect(reaches(cfg, block(cfg, 'b()'), block(cfg, 'use(y);'))).toBe(true); + const y = bindingIdx(cfg, 'y'); + expect(hasDef(cfg, y)).toBe(true); + expect(hasUse(cfg, y)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('expression-bodied member `=> k switch {…}` models the arms (#2207)', () => { + const cfg = cs.cfgOf(`class C { int G(int x) => x switch { 1 => a(), _ => b() }; }`); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + expect(edgeKinds(cfg).has('return')).toBe(true); + expect(reaches(cfg, block(cfg, 'a()'), cfg.exitIndex)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('assignment-RHS / single-arm value switch stays inline (documented gap)', () => { + const assign = cs.cfgOf( + `class C { int M(int x) { int y = 0; y = x switch { 1 => 1, _ => 2 }; return y; } }`, + ); + expect(edgeKinds(assign).has('switch-case')).toBe(false); + expect(reaches(assign, assign.entryIndex, assign.exitIndex)).toBe(true); + + const oneArm = cs.cfgOf(`class C { int M(int x) { var y = x switch { _ => 0 }; return y; } }`); + expect(edgeKinds(oneArm).has('switch-case')).toBe(false); + }); + + it('non-exhaustive switch expression (no `_` arm) keeps a no-match edge (EXIT reachable) (#2211)', () => { + const cfg = cs.cfgOf( + `class C { int M(int x) { var y = x switch { 1 => a(), 2 => b() }; return y; } }`, + ); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + // 2 arms + the conservative no-match path = 3 switch-case successors from the dispatch. + const dispatchIdx = block(cfg, 'x'); + expect(cfg.edges.filter((e) => e.from === dispatchIdx && e.kind === 'switch-case').length).toBe( + 3, + ); }); }); @@ -414,6 +467,18 @@ describe('C# CfgVisitor — does not throw on exotic shapes', () => { warn.mockRestore(); } }); + + it('a truncated value-position switch never throws out of the carrier path (R4) (#2211)', () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}); + try { + const root = cs.parse(`class C { int M(int x) { var y = x switch { 1 => a(`); + for (const fn of cs.collectFunctions(root)) { + expect(() => createCsharpCfgVisitor().buildFunctionCfg(fn, 'f.cs')).not.toThrow(); + } + } finally { + warn.mockRestore(); + } + }); }); // U6 — call-site `sites[]` taint substrate. INERT BY DESIGN: no C# taint model diff --git a/gitnexus/test/unit/cfg/dart-visitor.test.ts b/gitnexus/test/unit/cfg/dart-visitor.test.ts index 8fa1c225b..1e44ad463 100644 --- a/gitnexus/test/unit/cfg/dart-visitor.test.ts +++ b/gitnexus/test/unit/cfg/dart-visitor.test.ts @@ -71,6 +71,13 @@ describe('Dart CfgVisitor — structure', () => { expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); }); + it('a truncated value-position switch never throws out of the carrier path (R4) (#2211)', () => { + const root = dart.parse(`int f(int v){ var x = switch (v) { 1 => a(`); + for (const fn of dart.collectFunctions(root)) { + expect(() => createDartCfgVisitor().buildFunctionCfg(fn, 'f.dart')).not.toThrow(); + } + }); + it('a class method is a CFG-bearing function and binds its params', () => { const cfg = dart.cfgOf(`class C { void m(int a) { g(a); } }`); expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); @@ -258,30 +265,89 @@ describe('Dart CfgVisitor — switch', () => { expect(cfg.edges.some((e) => e.from === tainted && e.to === sink)).toBe(false); }); - it('a switch EXPRESSION used as a value stays inline (no branch edges)', () => { + it('value-position switch declaration is modeled as a dispatch, def bound at the join (#2207)', () => { const cfg = dart.cfgOf(`void f(int x) { - var y = switch (x) { 1 => one(), 2 => two(), _ => other() }; + var y = switch (x) { 1 => one(x), 2 => two(), _ => other() }; use(y); }`); - // The value-position switch expression coalesces; no switch-case edges. - expect(edgeKinds(cfg).has('switch-case')).toBe(false); - expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); - expect(definesBinding(cfg, bindingIdx(cfg, 'y'))).toBe(true); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + // each arm rejoins and reaches the downstream use of the bound result. + expect(reaches(cfg, block(cfg, 'one(x)'), block(cfg, 'use(y);'))).toBe(true); + expect(reaches(cfg, block(cfg, 'other()'), block(cfg, 'use(y);'))).toBe(true); + const y = bindingIdx(cfg, 'y'); + expect(definesBinding(cfg, y)).toBe(true); + expect(usesBinding(cfg, y)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); }); - it('a switch-EXPRESSION arm write is a may-def, not a hard kill of the prior def (#2206)', () => { + it('return switch (…) models each arm as returning the result (#2207)', () => { + const cfg = dart.cfgOf(`int f(int x) { + return switch (x) { 1 => a(x), 2 => b(), _ => c() }; + }`); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + expect(edgeKinds(cfg).has('return')).toBe(true); + expect(reaches(cfg, block(cfg, 'a(x)'), cfg.exitIndex)).toBe(true); + expect(reaches(cfg, block(cfg, 'c()'), cfg.exitIndex)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('a multi-binding decl with a switch-EXPRESSION value stays inline', () => { + const cfg = dart.cfgOf(`void f(int x) { + var y = switch (x) { _ => 0 }, z = 2; + use(y + z); + }`); + // Modeling a multi-binding decl arm-by-arm is out of scope — it coalesces. + expect(edgeKinds(cfg).has('switch-case')).toBe(false); + expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); + }); + + it('an INLINE switch-EXPRESSION arm write is a may-def, not a hard kill (#2206)', () => { + // An argument-position switch expression is NOT a modeled value-branch carrier + // (#2207 models only declaration / return), so it coalesces — and the harvest + // must still treat each arm write as a MAY-def (only one arm runs). const cfg = dart.cfgOf(`void f(int x) { int z = 0; - var y = switch (x) { 1 => z = 10, _ => z = 20 }; - use(z); + use(switch (x) { 1 => z = 10, _ => z = 20 }); + sink(z); }`); const z = bindingIdx(cfg, 'z'); - // only one arm runs, so the arm writes (z=10 / z=20) are MAY-defs — they must - // not unconditionally KILL the prior `int z = 0`. + expect(edgeKinds(cfg).has('switch-case')).toBe(false); expect(cfg.blocks.some((bl) => bl.statements?.some((s) => (s.mayDefs ?? []).includes(z)))).toBe( true, ); }); + + it('value switch without an unguarded `_` keeps the no-match edge (EXIT stays reachable) (#2211)', () => { + // A guarded `_ when …` is NOT an exhaustive catch-all — the conservative + // no-match path must remain (Dart throws at runtime if no arm + guard matches). + const cfg = dart.cfgOf(`int f(int v) { + return switch (v) { int n when n > 0 => a(n), _ when v < 0 => b() }; + }`); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + // the dispatch must reach the join WITHOUT going through an arm (the no-match edge). + const dispatchIdx = block(cfg, 'switch v'); + const dispatchSucc = cfg.edges.filter( + (e) => e.from === dispatchIdx && e.kind === 'switch-case', + ); + // dispatch fans to 2 arms + the no-match join = 3 switch-case successors. + expect(dispatchSucc.length).toBe(3); + }); + + it('a value-switch `when` guard is a conditional dispatch use, not an arm-value use (#2211)', () => { + const cfg = dart.cfgOf(`int f(int v) { + var x = switch (v) { int n when guardOk(v) => a(n), _ => b() }; + use(x); + }`); + const vIdx = bindingIdx(cfg, 'v'); + // `v` (used by the guard `guardOk(v)`) is recorded as a use on the dispatch + // block (text `switch v`), not buried in an arm-value block. + const dispatch = cfg.blocks.find((b) => b.text === 'switch v')!; + expect(dispatch.statements?.some((s) => s.uses.includes(vIdx))).toBe(true); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); }); describe('Dart CfgVisitor — try/on/catch/finally', () => { diff --git a/gitnexus/test/unit/cfg/java-visitor.test.ts b/gitnexus/test/unit/cfg/java-visitor.test.ts index 7070a9ee9..1cae93c04 100644 --- a/gitnexus/test/unit/cfg/java-visitor.test.ts +++ b/gitnexus/test/unit/cfg/java-visitor.test.ts @@ -294,13 +294,58 @@ describe('Java CfgVisitor — switch', () => { expect(hasUse(cfg, x)).toBe(true); }); - it('switch EXPRESSION value with yield stays inline; method has a single-exit CFG', () => { + it('value-position switch declaration is modeled as a dispatch, def bound at the join (#2207)', () => { const cfg = java.cfgOf(`class C { int m(int x) { int r = switch (x) { case 1 -> 10; default -> { yield 20; } }; - return r; + use(r); } }`); + // The arms are now real CFG blocks reached by switch-case dispatch edges. + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + // Each arm rejoins and reaches the use of the bound result. + expect(reaches(cfg, block(cfg, '10'), block(cfg, 'use(r);'))).toBe(true); + expect(reaches(cfg, block(cfg, 'yield 20;'), block(cfg, 'use(r);'))).toBe(true); + // `r` is defined (at the continuation) and used downstream — the chain is live. + const r = bindingIdx(cfg, 'r'); + expect(hasDef(cfg, r)).toBe(true); + expect(hasUse(cfg, r)).toBe(true); + // Modeling the arms yields control dependence (the whole point of #2207). + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); + }); + + it('return switch (…) {…} models each arm as returning the function result (#2207)', () => { + const cfg = java.cfgOf(`class C { int m(int x) { + return switch (x) { case 1 -> a(); case 2 -> b(); default -> c(); }; + } }`); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); expect(edgeKinds(cfg).has('return')).toBe(true); + // every arm reaches EXIT (its value IS the returned result). + expect(reaches(cfg, block(cfg, 'a()'), cfg.exitIndex)).toBe(true); + expect(reaches(cfg, block(cfg, 'c()'), cfg.exitIndex)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('value-position switch with ONE group stays inline (no real control dependence)', () => { + const cfg = java.cfgOf(`class C { int m(int x) { + int r = switch (x) { default -> 0; }; + use(r); + } }`); + // A single-arm switch carries no branch — it coalesces into the declaration block. + expect(edgeKinds(cfg).has('switch-case')).toBe(false); + expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); + }); + + it('assignment-RHS value switch stays inline (documented remaining gap)', () => { + const cfg = java.cfgOf(`class C { int m(int x) { + int r = 0; + r = switch (x) { case 1 -> 10; default -> 20; }; + use(r); + } }`); + // Only declaration / return carriers are modeled; an assignment RHS coalesces. + expect(edgeKinds(cfg).has('switch-case')).toBe(false); + expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); }); it('statement switch with a yield arm builds a dispatch with a yield block', () => { @@ -313,6 +358,48 @@ describe('Java CfgVisitor — switch', () => { // statement-position switch breaks a block → switch-case dispatch edges. expect(edgeKinds(cfg).has('switch-case')).toBe(true); }); + + it('colon-form value switch: yield ends the arm, NO fallthrough; every arm is CDG-dependent (#2211)', () => { + const cfg = java.cfgOf(`class C { int m(int k) { + int x = switch (k) { case 1: yield one(); case 2: yield two(); default: yield zero(); }; + use(x); + } }`); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + // a `yield` exits the switch — it does NOT fall through to the next colon group. + expect(edgeKinds(cfg).has('fallthrough')).toBe(false); + expect(reaches(cfg, block(cfg, 'one()'), block(cfg, 'two()'))).toBe(false); + // every arm rejoins and reaches the downstream use of the bound result. + expect(reaches(cfg, block(cfg, 'one()'), block(cfg, 'use(x);'))).toBe(true); + expect(reaches(cfg, block(cfg, 'two()'), block(cfg, 'use(x);'))).toBe(true); + // each arm is control-dependent on the dispatch — pin the SPECIFIC pairs. + const dispatch = block(cfg, 'k'); + const cdg = computeControlDependence(cfg); + expect( + cdg.edges.some( + (e) => e.controllerBlock === dispatch && e.dependentBlock === block(cfg, 'one()'), + ), + ).toBe(true); + expect( + cdg.edges.some( + (e) => e.controllerBlock === dispatch && e.dependentBlock === block(cfg, 'two()'), + ), + ).toBe(true); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('return switch (…) inside try/finally threads the finalizer per arm (#2211)', () => { + const cfg = java.cfgOf(`class C { int m(int k) { + try { + return switch (k) { case 1 -> a(); default -> b(); }; + } finally { cleanup(); } + } }`); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + // each arm's return threads the finally before EXIT. + expect(edgeKinds(cfg).has('finally-return')).toBe(true); + expect(reaches(cfg, block(cfg, 'a()'), block(cfg, 'cleanup();'))).toBe(true); + expect(reaches(cfg, block(cfg, 'b()'), block(cfg, 'cleanup();'))).toBe(true); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); }); describe('Java CfgVisitor — try / catch / finally / try-with-resources', () => { @@ -494,6 +581,18 @@ describe('Java CfgVisitor — does not throw on exotic shapes', () => { warn.mockRestore(); } }); + + it('a truncated value-position switch never throws out of the carrier path (R4) (#2211)', () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}); + try { + const root = java.parse(`class C { int m(int k){ int x = switch (k) { case 1 -> a(`); + for (const fn of java.collectFunctions(root)) { + expect(() => createJavaCfgVisitor().buildFunctionCfg(fn, 'f.java')).not.toThrow(); + } + } finally { + warn.mockRestore(); + } + }); }); // U6 — call-site `sites[]` taint substrate. INERT BY DESIGN: no Java taint model diff --git a/gitnexus/test/unit/cfg/kotlin-visitor.test.ts b/gitnexus/test/unit/cfg/kotlin-visitor.test.ts index 431565268..86efceeda 100644 --- a/gitnexus/test/unit/cfg/kotlin-visitor.test.ts +++ b/gitnexus/test/unit/cfg/kotlin-visitor.test.ts @@ -67,6 +67,13 @@ describe('Kotlin CfgVisitor — structure', () => { expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); }); + it('a truncated value-position when never throws out of the carrier path (R4) (#2211)', () => { + const root = kotlin.parse(`fun f(k: Int) { val x = when (k) { 0 ->`); + for (const fn of kotlin.collectFunctions(root)) { + expect(() => createKotlinCfgVisitor().buildFunctionCfg(fn, 'p.kt')).not.toThrow(); + } + }); + it('a class method is a CFG-bearing function', () => { const cfg = kotlin.cfgOf(`class C { fun m(a: Int) { g(a) } }`); expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); @@ -217,6 +224,78 @@ describe('Kotlin CfgVisitor — value-position branches (#2205)', () => { const cfg = kotlin.cfgOf(`fun f(k: Int) { val x = when (k) { else -> a() }; use(x) }`); expect(edgeKinds(cfg).has('switch-case')).toBe(false); }); + + it('x = when (...) assignment RHS models the arms; binds the target (#2205)', () => { + const cfg = kotlin.cfgOf( + `fun f(k: Int) { var x = 0; x = when (k) { 0 -> a(); else -> b() }; use(x) }`, + ); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + expect(reaches(cfg, block(cfg, 'a()'), block(cfg, 'use(x)'))).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(definesBinding(cfg, bindingIdx(cfg, 'x'))).toBe(true); + expect(usesBinding(cfg, bindingIdx(cfg, 'x'))).toBe(true); + }); + + it('x = if (c) ... else ... assignment RHS models both arms (#2205)', () => { + const cfg = kotlin.cfgOf(`fun f(c: Boolean) { var x = 0; x = if (c) a() else b(); use(x) }`); + expect(edgeKinds(cfg).has('cond-true')).toBe(true); + expect(edgeKinds(cfg).has('cond-false')).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(definesBinding(cfg, bindingIdx(cfg, 'x'))).toBe(true); + }); + + it('val x = try { ... } catch { ... } models the value-position try (#2205)', () => { + const cfg = kotlin.cfgOf( + `fun f() { val x = try { risky() } catch (e: Exception) { fallback() }; use(x) }`, + ); + // the try/catch is modeled as control flow (a throw edge to the handler)… + expect(edgeKinds(cfg).has('throw')).toBe(true); + // …and is CDG-bearing, with x bound at the rejoin and used downstream. + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(definesBinding(cfg, bindingIdx(cfg, 'x'))).toBe(true); + expect(usesBinding(cfg, bindingIdx(cfg, 'x'))).toBe(true); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('x = try { ... } catch { ... } assignment RHS models the value-position try (#2205)', () => { + const cfg = kotlin.cfgOf( + `fun f() { var x = 0; x = try { risky() } catch (e: Exception) { fallback() }; use(x) }`, + ); + expect(edgeKinds(cfg).has('throw')).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(definesBinding(cfg, bindingIdx(cfg, 'x'))).toBe(true); + expect(usesBinding(cfg, bindingIdx(cfg, 'x'))).toBe(true); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('return try { ... } catch { ... } models the value-position try; each arm returns (#2205, #2211)', () => { + const cfg = kotlin.cfgOf( + `fun f(): Int { return try { risky() } catch (e: Exception) { fallback() } }`, + ); + // the value-position try is modeled as control flow (throw edge to the handler)… + expect(edgeKinds(cfg).has('throw')).toBe(true); + expect(edgeKinds(cfg).has('return')).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('fun f() = try { ... } catch { ... } expression body models the value-position try (#2205, #2211)', () => { + const cfg = kotlin.cfgOf(`fun f(): Int = try { risky() } catch (e: Exception) { fallback() }`); + // visitExprBody routes the value-position try through control flow (throw edge), + // each arm yielding the function result (return), CDG-bearing. + expect(edgeKinds(cfg).has('throw')).toBe(true); + expect(edgeKinds(cfg).has('return')).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('a compound `x += ...` / a plain call RHS stays inline (not a value-branch carrier)', () => { + const compound = kotlin.cfgOf(`fun f(k: Int) { var x = 0; x += k; use(x) }`); + expect(edgeKinds(compound).has('switch-case')).toBe(false); + const call = kotlin.cfgOf(`fun f(k: Int) { var x = 0; x = compute(k); use(x) }`); + expect(edgeKinds(call).has('switch-case')).toBe(false); + expect(edgeKinds(call).has('cond-true')).toBe(false); + }); }); describe('Kotlin CfgVisitor — loops', () => { diff --git a/gitnexus/test/unit/cfg/php-visitor.test.ts b/gitnexus/test/unit/cfg/php-visitor.test.ts index 8afae6a82..3d3d5cd02 100644 --- a/gitnexus/test/unit/cfg/php-visitor.test.ts +++ b/gitnexus/test/unit/cfg/php-visitor.test.ts @@ -193,15 +193,51 @@ describe('PHP CfgVisitor — switch / match', () => { expect(isExitReachableFromAllBlocks(cfg)).toBe(true); }); - it('match is a value expression (no fallthrough), kept inline — value flows to the assign', () => { - const cfg = php.cfgOf(wrap(`$r = match ($x) { 1, 2 => "low", default => "high" }; return $r;`)); - // match arms are NOT separate dispatch blocks (documented inline-value gap). + it('value-position match assignment dispatches; target bound at the join (#2207)', () => { + const cfg = php.cfgOf( + wrap(`$r = match ($x) { 1, 2 => low($x), default => high() }; use_it($r);`), + ); + // arms dispatch as switch-case, never fall through. + expect(edgeKinds(cfg).has('switch-case')).toBe(true); expect(edgeKinds(cfg).has('fallthrough')).toBe(false); - expect(edgeKinds(cfg).has('switch-case')).toBe(false); - // The match value and the return both reach EXIT. - expect(reaches(cfg, block(cfg, 'match ($x)'), cfg.exitIndex)).toBe(true); + // each arm rejoins and reaches the downstream use of the bound result. + expect(reaches(cfg, block(cfg, 'low($x)'), block(cfg, 'use_it($r)'))).toBe(true); + expect(reaches(cfg, block(cfg, 'high()'), block(cfg, 'use_it($r)'))).toBe(true); + const r = bindingIdx(cfg, '$r'); + expect(hasDef(cfg, r)).toBe(true); + expect(hasUse(cfg, r)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); expect(isExitReachableFromAllBlocks(cfg)).toBe(true); }); + + it('return match (…) models each arm as returning the result (#2207)', () => { + const cfg = php.cfgOf(wrap(`return match ($x) { 1 => a($x), 2 => b(), default => c() };`)); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + expect(edgeKinds(cfg).has('return')).toBe(true); + expect(reaches(cfg, block(cfg, 'a($x)'), cfg.exitIndex)).toBe(true); + expect(reaches(cfg, block(cfg, 'c()'), cfg.exitIndex)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('single-arm match / ternary stays inline (no real control dependence)', () => { + const oneArm = php.cfgOf(wrap(`$r = match ($x) { default => 0 }; return $r;`)); + expect(edgeKinds(oneArm).has('switch-case')).toBe(false); + const ternary = php.cfgOf(wrap(`$r = $x > 0 ? a() : b(); return $r;`)); + expect(edgeKinds(ternary).has('switch-case')).toBe(false); + expect(isExitReachableFromAllBlocks(ternary)).toBe(true); + }); + + it('match without `default` keeps a no-match (UnhandledMatchError) edge; EXIT reachable (#2211)', () => { + const cfg = php.cfgOf(wrap(`$r = match ($x) { 1 => a($x), 2 => b() }; use_it($r);`)); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + // 2 arms + the conservative no-match path = 3 switch-case successors from the dispatch. + const dispatchIdx = block(cfg, '$x'); + expect(cfg.edges.filter((e) => e.from === dispatchIdx && e.kind === 'switch-case').length).toBe( + 3, + ); + }); }); describe('PHP CfgVisitor — try / catch / finally', () => { @@ -384,4 +420,11 @@ describe('PHP CfgVisitor — robustness', () => { expect(reachable(cfg, block(cfg, 'done()'))).toBe(true); expect(isExitReachableFromAllBlocks(cfg)).toBe(true); }); + + it('a truncated value-position match never throws out of the carrier path (R4) (#2211)', () => { + const root = php.parse(` a(`); + for (const fn of php.collectFunctions(root)) { + expect(() => createPhpCfgVisitor().buildFunctionCfg(fn, 'x.php')).not.toThrow(); + } + }); }); diff --git a/gitnexus/test/unit/cfg/swift-visitor.test.ts b/gitnexus/test/unit/cfg/swift-visitor.test.ts index 023596afc..a476c4eaa 100644 --- a/gitnexus/test/unit/cfg/swift-visitor.test.ts +++ b/gitnexus/test/unit/cfg/swift-visitor.test.ts @@ -1,6 +1,7 @@ import { describe, it, expect } from 'vitest'; import { requireVendoredGrammar } from '../../../src/core/tree-sitter/vendored-grammars.js'; import { createSwiftCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/swift.js'; +import type { FunctionCfg } from '../../../src/core/ingestion/cfg/types.js'; import { makeCfgHarness, type CfgHarness, @@ -54,6 +55,13 @@ describe('Swift CfgVisitor — structure', () => { expect(reaches(cfg, cfg.entryIndex, cfg.exitIndex)).toBe(true); }); + it('a truncated value-position if never throws out of the carrier path (R4) (#2211)', () => { + const root = swift.parse(`func f(v: Int) { let x = if v > 0 {`); + for (const fn of swift.collectFunctions(root)) { + expect(() => createSwiftCfgVisitor().buildFunctionCfg(fn, 'f.swift')).not.toThrow(); + } + }); + it('init and deinit are CFG-bearing functions', () => { const cfgs = swift.cfgsOf(`class C { init(x: Int) { self.x = x } ; deinit { cleanup() } }`); expect(cfgs).toHaveLength(2); @@ -229,6 +237,70 @@ describe('Swift CfgVisitor — switch (no implicit fallthrough)', () => { }); }); +describe('Swift CfgVisitor — value-position if/switch (Swift 5.9, #2207)', () => { + const hasDef = (cfg: FunctionCfg, idx: number): boolean => + cfg.blocks.some((bl) => bl.statements?.some((s) => s.defs.includes(idx))); + const hasUse = (cfg: FunctionCfg, idx: number): boolean => + cfg.blocks.some((bl) => bl.statements?.some((s) => s.uses.includes(idx))); + + it('`let x = if … else …` is modeled as a branch; def bound at the join', () => { + const cfg = swift.cfgOf(`func f(v: Int) { + let x = if v > 0 { hi() } else { lo() } + use(x) + }`); + expect(edgeKinds(cfg).has('cond-true')).toBe(true); + expect(edgeKinds(cfg).has('cond-false')).toBe(true); + expect(reaches(cfg, block(cfg, 'hi()'), block(cfg, 'use(x)'))).toBe(true); + expect(reaches(cfg, block(cfg, 'lo()'), block(cfg, 'use(x)'))).toBe(true); + const x = bindingIdx(cfg, 'x'); + expect(hasDef(cfg, x)).toBe(true); + expect(hasUse(cfg, x)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('`let y = switch v { … }` is modeled as a dispatch', () => { + const cfg = swift.cfgOf(`func f(v: Int) { + let y = switch v { case 1: one() ; default: other() } + use(y) + }`); + expect(edgeKinds(cfg).has('switch-case')).toBe(true); + expect(reaches(cfg, block(cfg, 'one()'), block(cfg, 'use(y)'))).toBe(true); + const y = bindingIdx(cfg, 'y'); + expect(hasDef(cfg, y)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('`return if … else …` models each arm as returning the result', () => { + const cfg = swift.cfgOf(`func f(v: Int) -> Int { + return if v > 0 { a() } else { b() } + }`); + expect(edgeKinds(cfg).has('cond-true')).toBe(true); + expect(edgeKinds(cfg).has('return')).toBe(true); + expect(reaches(cfg, block(cfg, 'a()'), cfg.exitIndex)).toBe(true); + expect(reaches(cfg, block(cfg, 'b()'), cfg.exitIndex)).toBe(true); + expect(computeControlDependence(cfg).edges.length).toBeGreaterThan(0); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('an else-less `if` value / plain binding stays inline (no real control dependence)', () => { + // `let x = g()` is a plain binding — no branch. + const cfg = swift.cfgOf(`func f(v: Int) { let x = g()\n use(x) }`); + expect(edgeKinds(cfg).has('cond-true')).toBe(false); + expect(edgeKinds(cfg).has('switch-case')).toBe(false); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('a single-entry value switch stays inline (below the >= 2 modeling threshold) (#2211)', () => { + // `isModelableValueBranch` requires >= 2 `switch_entry`; a one-entry value + // switch carries no real control dependence, so the decl coalesces inline. + const cfg = swift.cfgOf(`func f(v: Int) { let x = switch v { default: g() }\n use(x) }`); + expect(edgeKinds(cfg).has('switch-case')).toBe(false); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); +}); + describe('Swift CfgVisitor — do/catch (error handling)', () => { it('do/catch: a throw edge runs from each protected block to the handler', () => { const cfg = swift.cfgOf(`func f() { From cdb07289a4b239d738b4b5725d0092e40d68b030 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 15 Jun 2026 19:16:53 +0100 Subject: [PATCH 03/26] perf(cfg): SSA-sparse reaching-defs to replace the dense-set worklist (#2201) (#2212) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * test(cfg): retain dense reaching-defs as differential oracle + fuzz harness (#2201 U1) * refactor(cfg): extract shared harvest/adjacency/sweep + swappable in-set computer (#2201 U2) * perf(cfg): sparse change-driven reaching-defs solver + canonical truncation (#2201 U3,U4) * perf(cfg): switch production reaching-defs to the sparse solver (#2201 U5) * perf(cfg): true SSA-sparse reaching-defs solver with auto-dispatch (#2201 U3) Replace the per-variable worklist (correct but no faster — it still walks pass-through blocks per binding) with Cytron SSA: CHK dominators + dominance frontiers + phi-placement + stack renaming over a synthetic entry, answering block-entry reaching queries by walking the SSA def-use graph (SCC-condensed, cycle-safe). Pass-through blocks carry the dominating def via the rename stack and phi-nodes statically capture loop merges, so dense-bindings drops from O(n^2) to O(n) (5-23x faster, asymptotic) and deep nests are depth-independent. The sweep now queries a lazy reachingAt accessor with a sparse intra-block overlay (no full per-block lattice copy). Production auto-dispatches: SSA for looping functions >=16 blocks (where it pays off, incl. the deep nests the dense ceiling used to truncate -> ceiling stops firing), dense elsewhere (small / loop-free functions, 1.0x — no regression). Throw-edge and unreachable-block functions fall back to dense (byte-identical). Held byte-identical to the dense oracle across a 300k-CFG (~1.2M-comparison) differential fuzz. * test(cfg): R5 contrast — dense ceiling fires, SSA solver converges (#2201 U6) * bench(cfg): deep-nest scenario + tighten dense-bindings rd budget 10->2 (#2201 U7) dense-bindings rd_scaling drops 5.2->0.86 (SSA linear); budget tightened to 2.0. New deep-nest scenario (N nested loops, one carried var) measures rd under the production blocks×64 ceiling and asserts the SSA solver still COMPUTES full facts (facts_large_min) where the dense worklist would truncate — the ceiling-stops-firing acceptance. CFG fingerprints unchanged. * docs(cfg): document SSA-sparse solver + resolve the WTO no-go note (#2201 U8) * fix(review): apply autofix feedback (#2201) - Close the production SSA-dispatcher fuzz-coverage gap: the generator's maxBlocks=14 was below SSA_MIN_BLOCKS=16, so the auto-dispatcher's SSA branch was never differentially fuzzed. Raise to 36, add a hadLargeLoop coverage assertion + a back-edge-into-entry canonical CFG. Validated byte-identical on 100k random CFGs incl. >=16-block looping shapes via both entry points. - Correct stale function JSDocs + @internal annotations (dispatch/fallback roles). - Add an independent rd_all_computed bench gate (catches partial truncation). - maxBlockVisits comment, SSA_MIN_BLOCKS calibration note, nx->next rename. * fix(cfg): gate out-of-range binding indices to the dense fallback (#2201 review) Tri-review (adversarial lane, reproduced) found the SSA path less tolerant than the dense oracle it replaced: an out-of-range binding index in defs/uses/mayDefs (a corrupted/stale durable store) crashed the nBindings-sized arrays (defBlocks[v]/stacks[u]), where dense tolerated it as a Map key. The throw escaped the unguarded taint/harvest call sites and lost a whole file's taint layer. Add a malformed-input gate that falls back to the dense solver (which handles any index), preserving byte-identity AND the graceful per-function degradation. Add an OOB canonical CFG to the differential fuzz + a production- entry no-throw unit test (the generator only ever emitted in-range indices, so this divergent input was structurally invisible). * perf(cfg): bound the SSA value-graph, fall back to dense when oversized (#2201 review R1) maxFacts bounds fact materialization in sweepFacts, but nothing bounded the SSA-sparse solver's φ/value-graph construction. A high-binding-density deep loop routed to SSA (≥16 blocks + a reachable loop) builds an O(blocks×bindings) value graph the dense path would have truncated at its maxBlockVisits ceiling (~1.5 GB measured on a 3000-block × 300-binding function). Cap the value graph: after φ-placement (where nodeKeys.length == the φ count, the input-superlinear term) plus a 2×Σgen bound on the renaming nodes, fall back to computeInSetsDense before paying for renaming + Tarjan SCC. The fallback is byte-identical (dense is the equivalence oracle) and bounded (dense honors maxBlockVisits). Mirrors the existing throw/unreachable/OOB-binding gates. The ceiling is DEFAULT_MAX_SSA_VALUE_GRAPH_NODES (1e6 — far above any real or benchmarked function; dense-bindings/deep-nest build <1e4), overridable per call via ReachingDefsLimits.maxSsaValueGraphNodes. The new unit test makes the otherwise-invisible routing flip observable by pairing the cap with a tight maxBlockVisits (dense truncates, SSA computes). Equivalence fuzz unchanged (byte-identical, 20k CFGs green); tsc clean. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(cfg): alias single-source SCC reaching-sets in reachByScc (#2201 review R2) The SCC-condensation pass built a fresh Set for every SCC and copied each cross-SCC operand's reaching-set element-by-element — O(defs²) at wide-fan-in φ merges (a φ over many predecessors, each carrying a large reaching-set). Add an alias fast path: an SCC with no own leaf keys whose cross-SCC operands all resolve to ONE source SCC has exactly that source's reaching-set, so share it by reference instead of copying. This is the common shape (pass-through φ / single-operand value node). The full union is still built when an SCC has own keys or genuinely merges ≥2 distinct sources. Safe to share: reachByScc sets are read-only after construction (operand SCCs are numbered before s in Tarjan's reverse-topological order and are only iterated), and contents are identical — set iteration order is irrelevant because sweepFacts sorts each use's keys before emission (KTD6). Byte-identical to the dense oracle (30k-CFG fuzz green); tsc clean. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(cfg): fold the SSA reachability gate into the RPO pass (#2201 review R8) computeInSetsSparse ran a standalone reachability BFS to gate unreachable-block functions to the dense oracle, then immediately computed a reverse-post-order over the synthetic-entry graph — two traversals of the same successor structure. reversePostOrder now returns the reachability bitmap its DFS already builds, and the sparse path reuses it for the unreachable-block gate (S→entry is S's only edge, so reachX[b] for b * perf(cfg): trim per-statement/per-use/per-block allocations (#2201 review R9) Three transient allocations in the hot paths, all behavior-preserving: - sweepFacts: replace the per-statement `new Set([...defs, ...mayDefs])` with a direct `includes()` scan over the (1–3 element) def/mayDef arrays, guarded by a cheap hasSelfDefs flag that short-circuits pure-use statements. - sweepFacts: reuse a single scratch array for each use's reaching def-keys instead of spreading a fresh array per use. The KTD6 pre-sort still runs in place (load-bearing for truncated byte-identity). - computeInSetsSparse: build dPredsX by skipping consecutive-equal `from` values (preds[b] is pre-sorted by buildAdjacency, so duplicates are adjacent) instead of a per-block Set + spread + sort; the synthetic entry S = n exceeds every block index so it appends in order. The sweep is shared with the dense oracle, so these stay byte-identical on both paths — 50k-CFG fuzz (incl. maxFacts truncation, the order-sensitive case) green; tsc clean. Co-Authored-By: Claude Opus 4.8 (1M context) * docs(cfg): correct the sweepFacts truncation byte-identity mechanism (#2201 review R6) The outer sweepFacts JSDoc attributed a truncated result's cross-solver byte-identity to the two solvers producing "identical inSets — insertion order included". That is wrong: the dense (RPO fixpoint) and SSA (renaming/SCC) solvers deliberately build a loop-carried use's reaching set in DIFFERENT insertion orders — same set, different order. The actual mechanism is the KTD6 per-use sort that canonicalizes each use's keys by defKey BEFORE the maxFacts cutoff (already documented correctly on the inner comment). Rewrite the outer doc to say so. Documentation only. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(cfg): extract pure graph sub-stages to reaching-defs-graph.ts (#2201 review R4) reaching-defs.ts had grown to ~1190 lines with the #2201 SSA rewrite. Move the self-contained, pure (plain-array) algorithms into a sibling module: - reversePostOrder - buildDominators (Cooper-Harvey-Kennedy) - buildDominanceFrontiers (Cytron) - tarjanScc + condenseReachingSets (SCC condensation, alias fast path) - hasReachableLoop (dispatcher loop check) - unionSets / latticeEquals (def-set / lattice primitives) The new module has a STRICT one-way dependency (it imports nothing from reaching-defs.ts — every helper is parameterized over plain arrays/Sets), so there is no import cycle and each stage is independently testable. reaching-defs.ts now holds the orchestrator, the two solver bodies, harvest, adjacency, the statement sweep, and the dispatcher: 1190 → 988 lines. Pure mechanical extraction — behavior is preserved by the differential equivalence fuzz (40k CFGs byte-identical) + the reaching-defs unit/snapshot suites; tsc clean. The helpers are @internal (kept out of the shipped .d.ts by the stripInternal change). Co-Authored-By: Claude Opus 4.8 (1M context) * feat(pdg): stamp the reaching-defs solver identity for incremental re-analysis (#2201 review R3) The SSA-sparse rewrite computes full REACHING_DEF facts for deep-loop functions the old dense worklist truncated to empty at the blocks×64 ceiling. But an existing `--pdg` index carries those stale-truncated rows, and nothing forced a re-analysis: RepoMeta.pdg had no solver-identity key, so an upgraded run over an unchanged file kept the incremental fast path and never recomputed. Add a constant `reachingDefSolver: 'ssa-sparse-v1'` to the resolved pdg stamp (and to the RepoMeta['pdg'] type). It rides the existing key-union pdgModeMismatch comparator: a pre-#2201 stamp lacks the key, so 'ssa-sparse-v1' !== undefined trips one full writeback that recomputes the fuller coverage — no `--force` needed — exactly like the M2 REACHING_DEF cap and M5 CDG cap upgrade paths. A matching post-#2201 stamp compares equal, so there is no spurious re-analysis churn on steady-state re-runs. Tests: new pre-#2201→SSA upgrade block in pdg-mode-flip.test.ts (stamp present, absent-key mismatch, identical-stamp no-churn) + the persisted-stamp shape assertions and resolvePdgConfig DEFAULTS updated for the new key. tsc clean; pdg-mode-flip + run-analyze suites green (55/55). Co-Authored-By: Claude Opus 4.8 (1M context) * build(ts): stripInternal so @internal test-only exports stay out of the shipped .d.ts (#2201 review R5) computeReachingDefsDense/computeReachingDefsSparse are exported only for the equivalence fuzz and tagged @internal, but `declaration: true` emitted them into the public dist/**/*.d.ts. stripInternal removes any @internal-tagged export from the declaration output. This is repo-wide, which is the intended behavior: the same applies to every other test-only @internal export (hf-env's withDownloadTimeout etc., worker-pool's buildDispatchMessage/crashSignature, parse-impl's handleWorkerStartupFailure, the logger/safe-parse test resets, and the new reaching-defs-graph SSA helpers) — all of which are documented as not-public. Verified: - declaration emit succeeds with no TS4094/TS9006 ("cannot be named") errors; - the @internal functions are gone from the emitted .d.ts (reaching-defs-graph.d.ts is now `export {};`), while public symbols (computeReachingDefs) remain; - gitnexus-web — the only cross-package consumer — typechecks clean and imports only from gitnexus-shared, never from gitnexus internals; - runtime .js and the vitest/tsx tests are source-based, so unaffected. Co-Authored-By: Claude Opus 4.8 (1M context) * test(bench): add wide-merge scenario + tighten deep-nest facts floor (#2201 review R7) wide-merge: N bindings, each assigned in a 3-way branch (a wide multi-operand φ per binding) inside a loop, then all used after the merge. Unlike dense-bindings (one chained redef per `if`), every binding fans into its own wide φ, so the scenario exercises φ-placement + renaming + the reachByScc condensation across many independent wide merges. N bindings × constant arms ⇒ O(N) facts, so the gate is rd_scaling LINEARITY (measured ~1.07; budget 2.0 catches a regression to the per-binding-rescan O(N²) class the reachByScc alias path guards against). It runs the production SSA path (10007 blocks + a loop) and computes all facts under the blocks×64 budget (facts_large_min 24000 of a measured 26008 + the rd_all_computed gate). deep-nest: tighten facts_large_min 100 → 150 (measured 164) so a partial- truncation regression that still cleared 100 — but lost facts — now fails, with ~9% headroom for noise. bench --check PASS (9 scenarios) under --expose-gc; all existing CFG fingerprints unchanged. Co-Authored-By: Claude Opus 4.8 (1M context) * style(cfg): drop trailing blank line in reaching-defs.ts (prettier) Whitespace-only — a stray trailing newline left by the U4 extraction. `prettier --check` (the root format CI gate) now passes on every changed file. No behavior change. Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: Claude Opus 4.8 (1M context) --- gitnexus/bench/cfg/baselines.json | 21 +- gitnexus/bench/cfg/measure.mjs | 108 ++- gitnexus/src/core/ingestion/cfg/emit.ts | 14 +- .../core/ingestion/cfg/reaching-defs-graph.ts | 318 +++++++ .../src/core/ingestion/cfg/reaching-defs.ts | 843 ++++++++++++++---- gitnexus/src/core/run-analyze.ts | 8 + gitnexus/src/storage/repo-manager.ts | 13 + .../cfg/reaching-defs-equivalence.test.ts | 613 +++++++++++++ gitnexus/test/unit/cfg/reaching-defs.test.ts | 111 +++ gitnexus/test/unit/pdg-mode-flip.test.ts | 38 + gitnexus/test/unit/run-analyze.test.ts | 5 + gitnexus/tsconfig.json | 1 + 12 files changed, 1903 insertions(+), 190 deletions(-) create mode 100644 gitnexus/src/core/ingestion/cfg/reaching-defs-graph.ts create mode 100644 gitnexus/test/unit/cfg/reaching-defs-equivalence.test.ts diff --git a/gitnexus/bench/cfg/baselines.json b/gitnexus/bench/cfg/baselines.json index 606ccd19e..975a20426 100644 --- a/gitnexus/bench/cfg/baselines.json +++ b/gitnexus/bench/cfg/baselines.json @@ -29,8 +29,25 @@ "scaling_budget": 1.8, "disk_bytes_budget": 1.2, "heap_budget": 1.3, - "rd_scaling_budget": 10.0, - "_note": "#2082 M2: N bindings live across ~N blocks in one loop -- bindings x blocks scale JOINTLY (the solver-lattice stressor). The overlay design measures rd ~5.2 normalized: the OUT spine copy on genning blocks is O(V) per block, which is quadratic when V scales with B (bounded in prod by maxFunctionLines; real functions have V~10-40). Budget 10 deliberately tolerates that known shape and exists to catch the repo's recurring per-item-rescan class (a per-use scan over all defs is O(n^3) here, ratio >=16). If rd drops well below 5, tighten." + "rd_scaling_budget": 2.0, + "_note": "#2082 M2 / #2201 SSA: N bindings live across ~N blocks in one loop -- bindings x blocks scale JOINTLY (the solver-lattice stressor). The dense GEN/KILL worklist measured rd ~5.2 normalized here (the OUT spine copy is O(V) per block, quadratic when V scales with B). The #2201 SSA-sparse solver answers each use's reaching set from the def-use graph WITHOUT a per-block dense lattice, dropping rd to ~0.86 (linear; measured 5-23x faster absolute). Budget tightened 10->2: still absorbs noise + catches a regression to the per-item-rescan class (a per-use scan over all defs is O(n^3) here, ratio >=16), but now also catches a fall-back to the dense quadratic. Fingerprint unchanged -- CFG construction is untouched." + }, + "deep-nest": { + "fingerprint": "c0ca870487abc6ff379304c3162003e9e4f9b44aeb2fc29adfcf8d2179c7613a", + "scaling_budget": 1.8, + "disk_bytes_budget": 1.2, + "rd_scaling_budget": 2.0, + "facts_large_min": 150, + "_note": "#2201: N nested loops carrying ONE variable end-to-end (depth 40->160) -- the pathology the dense worklist is superlinear on and whose block-visit total drives it past the blocks×64 ceiling (it would TRUNCATE to empty). rd is measured under the PRODUCTION blocks×64 budget (rdProductionBudget) to prove the ceiling stops firing: the depth-INDEPENDENT SSA solver (phi-nodes capture loop merges statically; no fixpoint iteration) computes the full facts (measured 164 at large) with rd_scaling ~0.68 (linear in depth; measured ~0.57ms at depth 160). facts_large_min tightened 100->150 (#2201 review R7): a partial-truncation regression that still cleared the old floor of 100 (but lost facts of the measured 164) now fails, with ~9% headroom under 164 for noise; the companion rd_all_computed gate also catches any non-'computed' status. rd_scaling_budget 2.0 catches a regression back to superlinear. No heap_budget -- the deep-nest CFG payload is tiny and the retained-heap delta is GC-noise-dominated. Re-baseline the fingerprint only on an intentional CFG/visitor change." + }, + "wide-merge": { + "fingerprint": "7a66a844ee3994bd930c1e34bad3d7b410a762220e927c94fb3787f38d745280", + "scaling_budget": 1.8, + "disk_bytes_budget": 1.2, + "heap_budget": 1.3, + "rd_scaling_budget": 2.0, + "facts_large_min": 24000, + "_note": "#2201 review R7: N bindings, EACH assigned in a 3-way branch (a wide multi-operand phi per binding) inside a loop, then all used after the merge. Distinct from dense-bindings (one CHAINED redef per `if`): every binding fans into its OWN wide phi, so this exercises phi-placement + renaming + the reachByScc condensation across MANY independent wide merges. N bindings x constant arms => O(N) facts (measured 26008 at the large size), so the gate is rd_scaling LINEARITY: measured ~1.07 (time 9.3->39.8ms over the 4x size step); budget 2.0 catches a regression to the per-binding-rescan O(N^2) class -- the recurring solver antipattern the reachByScc alias fast path (review R2) guards against. rd is measured under the PRODUCTION blocks×64 budget (rdProductionBudget): all functions report 'computed' (the SSA path does not truncate here), and facts_large_min 24000 (measured 26008, ~7% headroom) + the rd_all_computed gate assert the wide merges compute fully. fp_blocks 82 / fp_edges 112 at FP_SIZE=15. Re-baseline the fingerprint only on an intentional CFG/harvest-shape change." }, "fact-fanout": { "fingerprint": "83a8243a8aff117f69aeecb39d02a483e6cca70439d75f63e433f4e4ac85578f", diff --git a/gitnexus/bench/cfg/measure.mjs b/gitnexus/bench/cfg/measure.mjs index efc0661a8..209afae55 100644 --- a/gitnexus/bench/cfg/measure.mjs +++ b/gitnexus/bench/cfg/measure.mjs @@ -44,7 +44,10 @@ import { fileURLToPath } from 'node:url'; import Parser from 'tree-sitter'; import { collectFunctionCfgs } from '../../src/core/ingestion/cfg/collect.ts'; import { computeReachingDefs } from '../../src/core/ingestion/cfg/reaching-defs.ts'; -import { DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION } from '../../src/core/ingestion/cfg/emit.ts'; +import { + DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, + DEFAULT_PDG_MAX_REACHING_DEF_BLOCK_REVISITS, +} from '../../src/core/ingestion/cfg/emit.ts'; import { getTreeSitterBufferSize } from '../../src/core/ingestion/constants.ts'; import { getLanguageGrammar } from '../../src/core/tree-sitter/parser-loader.ts'; import { getProvider } from '../../src/core/ingestion/languages/index.ts'; @@ -172,6 +175,56 @@ const SCENARIOS = [ return s + ' c = c - 1;\n }\n return v0;\n}\n'; }, }, + { + name: 'deep-nest', + // #2201: N nested loops carrying one variable end-to-end — the pathology the + // dense GEN/KILL worklist is superlinear on and that drives its block-visit + // total past the blocks×64 ceiling (it would truncate to an empty result). + // The production SSA solver is depth-INDEPENDENT (φ-nodes capture the loop + // merges statically; no fixpoint iteration), so rd time scales ~linearly + // with depth and the ceiling never fires. Two gates: rd_scaling_budget + // catches a regression back to superlinear, and facts_large_min asserts the + // solver still COMPUTES full facts under the PRODUCTION blocks×64 budget + // (rdProductionBudget) — a dense worklist would report zero facts here. + small: 40, + large: 160, // 4×, well under the visitor's recursive-nesting depth guard + rdMaxFacts: 0, // measure the algorithm, not the cap + rdProductionBudget: true, // pass blocks×64 — the SSA solver must still compute + gen: (n) => { + let s = 'function f(c: number) {\n let x = 0;\n'; + for (let i = 0; i < n; i++) s += ' '.repeat(i + 1) + `while (c > ${i}) {\n`; + s += ' '.repeat(n + 1) + 'x = x + 1;\n'; + for (let i = n - 1; i >= 0; i--) s += ' '.repeat(i + 1) + '}\n'; + return s + ' return x;\n}\n'; + }, + }, + { + name: 'wide-merge', + // #2201 review R7: N bindings, each assigned in a 3-way branch (a WIDE φ + // merge per binding) inside a loop, then all used after the merge. Unlike + // dense-bindings (one chained redef per `if`), every binding here fans into + // its own multi-operand φ — so the scenario stresses φ-placement + renaming + + // the reachByScc condensation across MANY independent wide merges. N bindings + // × constant arms ⇒ O(N) facts, so the gate is rd_scaling LINEARITY: a + // regression to the per-binding-rescan class (O(N²), the recurring solver + // antipattern reachByScc's alias fast path guards against) blows the ratio. + // >=16 blocks + a reachable loop ⇒ the production SSA path. + rdMaxFacts: 0, // measure the algorithm, not the cap + rdProductionBudget: true, // prove the SSA path computes under blocks×64 + gen: (n) => { + let s = 'function f(c: number) {\n'; + for (let i = 0; i < n; i++) s += ` let v${i} = ${i};\n`; + s += ' while (c > 0) {\n'; + for (let i = 0; i < n; i++) { + s += + ` if (c > ${i}) { v${i} = ${i} + c; }` + + ` else if (c < ${i}) { v${i} = ${i} - c; }` + + ` else { v${i} = c; }\n`; + } + for (let i = 0; i < n; i++) s += ` use(v${i});\n`; + return s + ' c = c - 1;\n }\n return v0;\n}\n'; + }, + }, { name: 'fact-fanout', // #2082 M2: N parallel case-arm defs of one variable + N later uses — @@ -296,17 +349,32 @@ function measureCollect(tk, src, file, reps) { // the scope-resolution emit loop adds per file on a --pdg run). `maxFacts` // mirrors the per-scenario production posture: 0 (unlimited) measures the // algorithm; the production default exercises the boundedness contract. -function measureReachingDefs(cfgs, reps, maxFacts) { - for (const c of cfgs) computeReachingDefs(c, { maxFacts }); // warm JIT +// When `blockVisitsMul` > 0 each call also passes the PRODUCTION per-function +// maxBlockVisits budget (blocks × mul). On the deep-nest scenario this is how +// "the ceiling stops firing" (#2201) is measured: the dense worklist would +// truncate to an empty result under this budget, whereas the production SSA +// solver computes the full facts — so a nonzero `facts` under the budget is the +// gate (see facts_large_min in baselines.json). +function measureReachingDefs(cfgs, reps, maxFacts, blockVisitsMul = 0) { + const limitsFor = (c) => + blockVisitsMul > 0 + ? { maxFacts, maxBlockVisits: c.blocks.length * blockVisitsMul } + : { maxFacts }; + for (const c of cfgs) computeReachingDefs(c, limitsFor(c)); // warm JIT const samples = []; let facts = 0; + let allComputed = true; for (let i = 0; i < reps; i++) { const start = process.hrtime.bigint(); facts = 0; - for (const c of cfgs) facts += computeReachingDefs(c, { maxFacts }).facts.length; + for (const c of cfgs) { + const r = computeReachingDefs(c, limitsFor(c)); + facts += r.facts.length; + if (r.status !== 'computed') allComputed = false; + } samples.push(Number(process.hrtime.bigint() - start) / 1e6); } - return { ms: median(samples), facts }; + return { ms: median(samples), facts, allComputed }; } // ---- taint pass cost (#2083 M3 U7) ---- @@ -441,10 +509,14 @@ function measureScenario(scenario) { ? heapLarge / heapSmall / sizeRatio : null; - // #2082 M2: reaching-defs solve cost over the same CFGs. + // #2082 M2: reaching-defs solve cost over the same CFGs. #2201: scenarios + // marked `rdProductionBudget` also pass the per-function blocks×64 ceiling, to + // prove the production SSA solver still COMPUTES where the dense worklist would + // truncate (the deep-nest ceiling-stops-firing acceptance). const rdMaxFacts = scenario.rdMaxFacts ?? 0; - const rdSmall = measureReachingDefs(small.cfgs, REPS, rdMaxFacts); - const rdLarge = measureReachingDefs(large.cfgs, REPS, rdMaxFacts); + const rdBudgetMul = scenario.rdProductionBudget ? DEFAULT_PDG_MAX_REACHING_DEF_BLOCK_REVISITS : 0; + const rdSmall = measureReachingDefs(small.cfgs, REPS, rdMaxFacts, rdBudgetMul); + const rdLarge = measureReachingDefs(large.cfgs, REPS, rdMaxFacts, rdBudgetMul); // Clamp the denominator: a 0.000ms small-N median would otherwise yield // ratio 0 and the gate would self-disable exactly when the solver is fast. const rdRatio = rdLarge.ms / Math.max(rdSmall.ms, 0.001) / sizeRatio; @@ -506,6 +578,7 @@ function measureScenario(scenario) { rd_scaling_ratio: Number(rdRatio.toFixed(3)), facts_small: rdSmall.facts, facts_large: rdLarge.facts, + rd_all_computed: rdLarge.allComputed, ...fingerprint(tk, scenario), }; } @@ -570,6 +643,25 @@ if (!CHECK) { `(the maxFacts early-stop is the boundedness contract)`, ); } + // #2201 deep-nest: under the PRODUCTION blocks×64 budget the SSA solver must + // still COMPUTE full facts (a nonzero floor) where the dense worklist would + // truncate to empty — "the ceiling stops firing". + if (base.facts_large_min !== undefined && r.facts_large < base.facts_large_min) { + failures.push( + `${r.scenario}: only ${r.facts_large} facts < floor ${base.facts_large_min} under the ` + + `production block-visit budget — the ceiling fired (SSA should not truncate here)` + + (r.rd_all_computed ? '' : ` [status != computed]`), + ); + } + // Independent of the fact-count floor: under the production budget every + // function in a facts_large_min scenario must report status 'computed'. This + // catches a partial-truncation regression that still clears the count floor. + if (base.facts_large_min !== undefined && r.rd_all_computed === false) { + failures.push( + `${r.scenario}: a function did not reach status 'computed' under the production ` + + `block-visit budget — the SSA solver truncated where it must compute`, + ); + } if (base.disk_bytes_large_max !== undefined && r.disk_bytes_large > base.disk_bytes_large_max) { failures.push( `${r.scenario}: cfgSideChannel absolute size ${r.disk_bytes_large} > ceiling ` + diff --git a/gitnexus/src/core/ingestion/cfg/emit.ts b/gitnexus/src/core/ingestion/cfg/emit.ts index 998f4c79e..f4550951d 100644 --- a/gitnexus/src/core/ingestion/cfg/emit.ts +++ b/gitnexus/src/core/ingestion/cfg/emit.ts @@ -108,10 +108,16 @@ export const DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION = * never fires). Truncation degrades to a sound empty REACHING_DEF for that one * function (status `truncated`), never wrong facts. * - * This ceiling is the SOUND backstop, not a perf fix: WTO / loop-aware iteration - * ordering was benchmarked and rejected (0% faster — the cost is dense-set - * propagation, not visitation order; see the no-go note in reaching-defs.ts at - * the RPO-order site). SSA-sparse reaching-defs is the deferred real fix. + * As of #2201 this ceiling is an adversarial-only backstop that effectively + * never fires on real code: the production solver auto-selects the SSA-sparse + * path for the looping functions that would breach it, and the SSA path has no + * fixpoint iteration (it answers reaching queries from the def-use graph in one + * pass) so it computes the full facts where the dense worklist would have + * truncated. The budget is still consulted on the dense fallback path (small / + * loop-free functions, and throw-edge / unreachable-block functions the SSA path + * does not model). WTO / loop-aware iteration ordering was benchmarked and + * rejected (0% faster — the cost was dense-set propagation, not visitation + * order); SSA-sparse was the real fix. See reaching-defs.ts. */ export const DEFAULT_PDG_MAX_REACHING_DEF_BLOCK_REVISITS = 64; diff --git a/gitnexus/src/core/ingestion/cfg/reaching-defs-graph.ts b/gitnexus/src/core/ingestion/cfg/reaching-defs-graph.ts new file mode 100644 index 000000000..5088b2770 --- /dev/null +++ b/gitnexus/src/core/ingestion/cfg/reaching-defs-graph.ts @@ -0,0 +1,318 @@ +/** + * Pure graph sub-stages for the reaching-definitions solvers (#2201 review R4). + * + * Extracted from reaching-defs.ts to keep that module focused on the + * orchestrator, the dense oracle, the statement sweep, and the dispatcher. + * Everything here is a pure function of plain arrays — no CFG, no harvest, no + * solver state — so this module has NO dependency on reaching-defs.ts (a strict + * one-way import) and each stage is independently testable. The SSA pipeline + * (dominators → dominance frontiers → Tarjan SCC → reach-set condensation) + * implements Cooper-Harvey-Kennedy + Cytron + Tarjan; reverse-post-order, the + * loop-reachability check, and the def-set/lattice primitives are shared with + * the dense GEN/KILL solver and the dispatcher. + * + * These are held byte-identical to their former inline form by the differential + * equivalence fuzz (test/unit/cfg/reaching-defs-equivalence.test.ts) — any diff + * after extraction is an extraction bug, never the oracle. + */ + +/** def-site keys reaching a program point (see reaching-defs.ts). */ +type DefSet = Set; +/** bindingIdx → def-site keys (the dense solver's per-block lattice). */ +type Lattice = Map; + +/** + * RPO over blocks reachable from `entry`; unreachable blocks appended by index. + * Returns the order AND the reachability bitmap the DFS already computed, so a + * caller needing "is every block reachable?" reuses this pass instead of a + * separate BFS (#2201 review R8 — the SSA path's reachability gate). + * + * @internal + */ +export function reversePostOrder( + entry: number, + succs: readonly number[][], + n: number, +): { order: number[]; visited: boolean[] } { + const visited = new Array(n).fill(false); + const post: number[] = []; + // Iterative DFS with an explicit phase stack (children pushed in reverse so + // they pop in sorted order — determinism). + const stack: { node: number; childIdx: number }[] = [{ node: entry, childIdx: 0 }]; + visited[entry] = true; + while (stack.length) { + const top = stack[stack.length - 1]; + const children = succs[top.node]; + if (top.childIdx < children.length) { + const next = children[top.childIdx]; + top.childIdx += 1; + if (!visited[next]) { + visited[next] = true; + stack.push({ node: next, childIdx: 0 }); + } + } else { + post.push(top.node); + stack.pop(); + } + } + const order = post.reverse(); + for (let b = 0; b < n; b++) if (!visited[b]) order.push(b); + return { order, visited }; +} + +/** + * Immediate dominators (Cooper-Harvey-Kennedy; correct on irreducible CFGs). + * `rpo` is the reverse-post-order rooted at the synthetic start `S`, `dPredsX` + * the dominator-graph predecessors (incl. S→entry). Returns idom[b] for every + * node in [0, nx); idom[S] === S. + * + * @internal + */ +export function buildDominators( + rpo: readonly number[], + dPredsX: readonly number[][], + S: number, + nx: number, +): number[] { + const rpoIdx = new Array(nx); + rpo.forEach((b, i) => (rpoIdx[b] = i)); + const idom = new Array(nx).fill(-1); + idom[S] = S; + const intersect = (a: number, b: number): number => { + while (a !== b) { + while (rpoIdx[a] > rpoIdx[b]) a = idom[a]; + while (rpoIdx[b] > rpoIdx[a]) b = idom[b]; + } + return a; + }; + for (let changed = true; changed; ) { + changed = false; + for (const b of rpo) { + if (b === S) continue; + let nd = -1; + for (const p of dPredsX[b]) if (idom[p] !== -1) nd = nd === -1 ? p : intersect(nd, p); + if (nd !== -1 && idom[b] !== nd) { + idom[b] = nd; + changed = true; + } + } + } + return idom; +} + +/** + * Dominance frontiers (Cytron). df[b] is the set of nodes where b's dominance + * ends — the φ-placement targets for any binding defined in b. + * + * @internal + */ +export function buildDominanceFrontiers( + dPredsX: readonly number[][], + idom: readonly number[], + nx: number, +): Set[] { + const df: Set[] = Array.from({ length: nx }, () => new Set()); + for (let b = 0; b < nx; b++) { + const dp = dPredsX[b]; + if (dp.length < 2) continue; + for (const p of dp) { + let runner = p; + while (runner !== idom[b] && runner !== -1) { + df[runner].add(b); + runner = idom[runner]; + } + } + } + return df; +} + +/** + * Tarjan strongly-connected components over the value-graph operand edges + * (`nodeOps[node]` = operand node ids). Iterative (explicit work stack — the + * graph can be deep). SCCs are emitted in REVERSE topological order, so an + * SCC's operand SCCs are numbered before it — the property + * {@link condenseReachingSets} relies on for its single forward pass. + * + * @internal + */ +export function tarjanScc(nodeOps: readonly number[][]): { + sccOf: number[]; + sccMembers: number[][]; +} { + const N = nodeOps.length; + const sccOf = new Array(N).fill(-1); + const sccMembers: number[][] = []; + const index = new Array(N).fill(-1); + const low = new Array(N).fill(0); + const onStk = new Array(N).fill(false); + const tarjanStk: number[] = []; + let counter = 0; + for (let start = 0; start < N; start++) { + if (index[start] !== -1) continue; + const work: { node: number; oi: number }[] = [{ node: start, oi: 0 }]; + index[start] = low[start] = counter++; + tarjanStk.push(start); + onStk[start] = true; + while (work.length) { + const top = work[work.length - 1]; + const ops = nodeOps[top.node]; + if (top.oi < ops.length) { + const w = ops[top.oi++]; + if (index[w] === -1) { + index[w] = low[w] = counter++; + tarjanStk.push(w); + onStk[w] = true; + work.push({ node: w, oi: 0 }); + } else if (onStk[w] && index[w] < low[top.node]) { + low[top.node] = index[w]; + } + } else { + if (low[top.node] === index[top.node]) { + const members: number[] = []; + let w: number; + do { + w = tarjanStk.pop()!; + onStk[w] = false; + sccOf[w] = sccMembers.length; + members.push(w); + } while (w !== top.node); + sccMembers.push(members); + } + work.pop(); + if (work.length) { + const par = work[work.length - 1].node; + if (low[top.node] < low[par]) low[par] = low[top.node]; + } + } + } + } + return { sccOf, sccMembers }; +} + +/** + * Reaching def-key set per SCC via condensation (cycle-safe union). Tarjan emits + * SCCs in reverse topological order, so a single forward pass over SCCs resolves + * every union: an SCC's reaching set is its members' own leaf keys plus the + * already-computed reaching sets of its cross-SCC operands. + * + * Alias fast path (#2201 review R2): an SCC with NO own leaf keys whose cross-SCC + * operands all resolve to a SINGLE source SCC has exactly that source's reaching + * set — share it BY REFERENCE instead of copying element-by-element (the O(defs²) + * cost at wide-fan-in φ merges). Safe: the returned sets are read-only after this + * pass, and contents are identical (set iteration order is irrelevant — the + * sweep sorts each use's keys before emission, KTD6). + * + * @internal + */ +export function condenseReachingSets( + sccMembers: readonly number[][], + sccOf: readonly number[], + nodeKeys: readonly (DefSet | null)[], + nodeOps: readonly number[][], +): DefSet[] { + const reachByScc: DefSet[] = new Array(sccMembers.length); + for (let s = 0; s < sccMembers.length; s++) { + const members = sccMembers[s]; + let aliasTarget = -1; // the unique cross-SCC source SCC, or -1 if none/many + let hasOwnKeys = false; + let multiSource = false; + for (const node of members) { + if (nodeKeys[node]) { + hasOwnKeys = true; + break; + } + for (const w of nodeOps[node]) { + const ws = sccOf[w]; + if (ws === s) continue; // intra-SCC operand: same set being built, adds nothing + if (aliasTarget === -1) aliasTarget = ws; + else if (aliasTarget !== ws) { + multiSource = true; + break; + } + } + if (multiSource) break; + } + if (!hasOwnKeys && !multiSource && aliasTarget !== -1) { + reachByScc[s] = reachByScc[aliasTarget]; // zero-copy share + continue; + } + // General case: union own leaf keys + every distinct cross-SCC operand set. + const set: DefSet = new Set(); + for (const node of members) { + const keys = nodeKeys[node]; + if (keys) for (const k of keys) set.add(k); + for (const w of nodeOps[node]) { + const ws = sccOf[w]; + if (ws !== s) for (const k of reachByScc[ws]) set.add(k); + } + } + reachByScc[s] = set; + } + return reachByScc; +} + +/** + * True iff a cycle is reachable from `entry` (the CFG has a loop). Iterative DFS + * with a gray/black coloring; a gray successor is a back-edge. O(V+E). Used by + * the production dispatcher to decide SSA-vs-dense. + * + * @internal + */ +export function hasReachableLoop(entry: number, succs: readonly number[][], n: number): boolean { + const color = new Uint8Array(n); // 0 white, 1 gray, 2 black + const stack: { node: number; i: number }[] = [{ node: entry, i: 0 }]; + color[entry] = 1; + while (stack.length) { + const top = stack[stack.length - 1]; + const ss = succs[top.node]; + if (top.i < ss.length) { + const next = ss[top.i++]; + if (color[next] === 1) return true; + if (color[next] === 0) { + color[next] = 1; + stack.push({ node: next, i: 0 }); + } + } else { + color[top.node] = 2; + stack.pop(); + } + } + return false; +} + +/** + * Order-stable union of two def-sets (shares `a` when `b` adds nothing). + * + * @internal + */ +export function unionSets(a: DefSet, b: DefSet): DefSet { + let target = a; + let copied = false; + for (const key of b) { + if (!target.has(key)) { + if (!copied) { + target = new Set(a); + copied = true; + } + target.add(key); + } + } + return target; +} + +/** + * Per-binding lattice equality with a reference fast path (sets only ever grow). + * + * @internal + */ +export function latticeEquals(a: Lattice, b: Lattice): boolean { + if (a === b) return true; + if (a.size !== b.size) return false; + for (const [k, bSet] of b) { + const aSet = a.get(k); + if (aSet === bSet) continue; + if (!aSet || aSet.size !== bSet.size) return false; + for (const v of bSet) if (!aSet.has(v)) return false; + } + return true; +} diff --git a/gitnexus/src/core/ingestion/cfg/reaching-defs.ts b/gitnexus/src/core/ingestion/cfg/reaching-defs.ts index ff192e802..4e9f3c23b 100644 --- a/gitnexus/src/core/ingestion/cfg/reaching-defs.ts +++ b/gitnexus/src/core/ingestion/cfg/reaching-defs.ts @@ -1,8 +1,27 @@ /** - * Reaching definitions (#2082 M2 U3) — classic GEN/KILL monotone fixpoint over - * one function's CFG, plus the canonical intra-block statement sweep that - * recovers statement-granular def→use facts from M1's coalesced blocks - * WITHOUT re-splitting the CFG. + * Reaching definitions (#2082 M2 U3, SSA-sparse rewrite #2201) — per-function + * intraprocedural may-reaching-definitions, plus the canonical intra-block + * statement sweep that recovers statement-granular def→use facts from M1's + * coalesced blocks WITHOUT re-splitting the CFG. + * + * ARCHITECTURE (#2201): the analysis is split into solver-INDEPENDENT stages + * (shared by every path, so the byte-identical surface is maximal) and a + * swappable IN-set computation: + * - {@link harvestStatementFacts} — per-block GEN/allDefs + def/use telemetry. + * - {@link buildAdjacency} — throw-aware predecessor/successor adjacency. + * - the IN-set computer — answers block-entry reaching-set queries. Two + * implementations: {@link computeInSetsSparse} (SSA — CHK dominators → + * Cytron dominance frontiers + φ-placement → stack renaming over a + * synthetic entry, walked SCC-condensed) and {@link computeInSetsDense} + * (the original GEN/KILL worklist). Production runs {@link + * computeInSetsAuto}, which picks the SSA solver for looping functions large + * enough to amortize construction (where it is asymptotically faster and + * never hits the dense ceiling) and the dense worklist everywhere else; the + * dense path also serves the throw-edge / unreachable-block cases the SSA + * path does not model. The two are held byte-identical by the equivalence + * fuzz — only set CONTENTS must match (the sweep sorts each use's keys + * before the maxFacts cutoff, so iteration order is irrelevant). + * - {@link sweepFacts} — statement sweep + sort + maxFacts truncation. * * PURE AND DETERMINISTIC (load-bearing contract): * - Pure function of its inputs — no graph, no logger (warnings are the @@ -14,17 +33,10 @@ * insertion-ordered Maps/Sets throughout, and the output fact array is * explicitly sorted. Snapshot tests and content-derived edge ids rely on it. * - * COMPLEXITY DISCIPLINE (the four-times-repeated repo bug shape is per-item - * re-derivation inside the loop): def-sets are SHARED BY REFERENCE, never - * deep-copied — a MUST def's kill is total per binding, so a transfer either - * aliases the incoming set or replaces it; a MAY def (conditional context — - * see StatementFacts.mayDefs) unions WITHOUT killing via a copy-on-extend. - * Single-predecessor blocks alias the predecessor's OUT map outright; - * multi-pred merges union only bindings whose incoming sets differ by - * reference. Iteration is reverse post-order, seeded with every block - * (unreachable blocks keep ⊥ IN — correct, their defs reach nothing). - * Convergence: sets grow monotonically within the finite def-site universe ⇒ - * ≤ loop-depth+1 passes in practice. + * COMPLEXITY DISCIPLINE: def-sets are SHARED BY REFERENCE, never deep-copied — + * a MUST def's kill is total per binding, so a transfer either aliases the + * incoming set or replaces it; a MAY def (conditional context — see + * StatementFacts.mayDefs) unions WITHOUT killing via a copy-on-extend. * * `limits.maxFacts` bounds materialization: facts are O(defs×uses) BY SPEC in * merge-heavy code (N branch-arm defs × N later uses = N² facts), and a @@ -34,6 +46,16 @@ * as a per-function taint-coverage gap. */ import type { BindingEntry, FunctionCfg } from './types.js'; +import { + buildDominanceFrontiers, + buildDominators, + condenseReachingSets, + hasReachableLoop, + latticeEquals, + reversePostOrder, + tarjanScc, + unionSets, +} from './reaching-defs-graph.js'; /** A statement-granular program point within one function's CFG. */ export interface ProgramPoint { @@ -68,18 +90,42 @@ export interface ReachingDefsLimits { */ readonly maxFacts?: number; /** - * Maximum total block dequeues in the dataflow fixpoint. Iterative - * reaching-defs on a reducible CFG converges in O(loop-nesting-depth) passes, - * so a worklist visits each block a small multiple of times for real code; a - * pathologically deep loop nest (machine-generated / obfuscated) drives the - * pass count — and thus the visit total — to O(blocks²) and the solver to - * seconds + GB of heap (`maxFacts` does not help: fact count stays linear). - * When the visit total exceeds this budget the fixpoint has NOT converged, so - * any facts would be unsound — the solver bails to a sound empty - * `status: 'truncated'` (like the `overflow` guard). `undefined`/0 ⇒ unlimited - * (the default for direct callers; the emit path sets a per-function budget). + * Adversarial-only safety bound on the DENSE worklist's iteration. + * + * The dense GEN/KILL solver reads this as a ceiling on total block dequeues: + * iterative reaching-defs on a reducible CFG converges in O(loop-nesting-depth) + * passes, but a pathologically deep loop nest drives the visit total — and thus + * the solver — to O(blocks²), seconds + GB of heap (`maxFacts` does not help: + * fact count stays linear). Exceeding the budget means the fixpoint has NOT + * converged, so any facts would be unsound — the dense solver bails to a sound + * empty `status: 'truncated'` (like the `overflow` guard). + * + * The SSA solver (#2201) has NO fixpoint iteration — it answers reaching + * queries from the def-use graph in one pass — so it always converges and this + * budget never trips it. The production dispatcher ({@link computeInSetsAuto}) + * routes the deep nests that would breach the dense ceiling to the SSA solver, + * which computes their full facts: the ceiling that fired on the dense worklist + * effectively never fires on real code (#2201 acceptance). The budget is still + * honored on the dense fallback path (small / loop-free functions, and the + * throw-edge / unreachable-block cases the SSA path does not model). + * + * `undefined`/0 ⇒ unlimited (the default for direct callers; the emit path sets + * a per-function budget). */ readonly maxBlockVisits?: number; + /** + * Memory bound on the SSA-sparse solver's value-graph construction (#2201 + * review R1). `maxFacts` bounds fact MATERIALIZATION (sweepFacts) but nothing + * bounds the φ/value-graph the sparse path builds first; a high-binding-density + * deep loop routed to SSA (≥ SSA_MIN_BLOCKS blocks + a reachable loop) builds an + * O(blocks×bindings) graph the dense path would have truncated at the + * `maxBlockVisits` ceiling (~1.5 GB measured on a 3000-block × 300-binding + * function). When the projected node count would exceed this, the sparse solver + * falls back to the dense oracle (byte-identical, and bounded — dense honors + * `maxBlockVisits`). Honored ONLY by the sparse path; the dense solver ignores + * it. `undefined`/0 ⇒ {@link DEFAULT_MAX_SSA_VALUE_GRAPH_NODES}. + */ + readonly maxSsaValueGraphNodes?: number; } export interface FunctionDefUse { @@ -111,7 +157,7 @@ export interface FunctionDefUse { * statements into one block, so an overflow would silently alias * (block b, stmt STRIDE+k) with (block b+1, stmt k) and fabricate wrong-block * facts. computeReachingDefs therefore range-checks up front and bails to a - * sound empty `truncated` result instead of ever letting a key alias. + * sound empty `overflow` result instead of ever letting a key alias. * 2^21 statements per block × blocks ≤ 2^32 stays inside Number's 2^53. */ const STMT_STRIDE = 1 << 21; @@ -124,11 +170,125 @@ type Lattice = Map; const EMPTY_LATTICE: Lattice = new Map(); +/** A block's GEN entry for one binding: the genned set + whether it kills. */ +interface GenEntry { + set: DefSet; + kills: boolean; +} + +/** Solver-independent per-block facts (shared by both IN-set computers). */ +interface Harvest { + /** gen[b]: bindingIdx → { set, kills }. A MUST def kills; a MAY def adds. */ + readonly gen: readonly (Map | null)[]; + /** allDefsGen[b]: bindingIdx → EVERY def-site key in the block (throw edges). */ + readonly allDefsGen: readonly (Lattice | null)[]; + readonly defLine: ReadonlyMap; + readonly defCount: number; + readonly useCount: number; +} + +/** Throw-aware adjacency (shared by both IN-set computers). */ +interface Adjacency { + readonly preds: readonly { from: number; viaThrow: boolean }[][]; + readonly succs: readonly number[][]; + /** Handlers whose IN depends on a block's IN (throw edges). */ + readonly throwSuccs: readonly number[][]; +} + +/** + * Block-entry reaching-set accessor: the set of def-site keys of `binding` + * reaching `blockIndex`'s entry, or undefined when none reach. Both solvers + * expose their result through this accessor so the sweep is solver-agnostic; + * the dense oracle backs it with precomputed per-block lattices, the sparse + * solver computes it lazily from the SSA def-use graph. Because {@link + * sweepFacts} sorts each use's reaching keys before the maxFacts cutoff, only + * the set CONTENTS need to match across solvers — not iteration order. + */ +type ReachingAt = (blockIndex: number, binding: number) => DefSet | undefined; + +/** + * The swappable stage: a block-entry reaching-set accessor, or a non- + * convergence signal (the work budget exceeded ⇒ sound empty `truncated`). + */ +type InSetsResult = { converged: true; reachingAt: ReachingAt } | { converged: false }; + +type InSetsComputer = ( + cfg: FunctionCfg, + n: number, + h: Harvest, + adj: Adjacency, + limits: ReachingDefsLimits | undefined, +) => InSetsResult; + /** * Compute reaching definitions for one function. See the module doc for the * purity/determinism/sharing contract. + * + * This is the production entry point. As of #2201 it auto-dispatches via + * {@link computeInSetsAuto} — the SSA-sparse solver ({@link computeInSetsSparse}) + * for looping functions large enough to amortize construction, the dense + * GEN/KILL worklist ({@link computeInSetsDense}) everywhere else (and for the + * throw-edge / unreachable-block functions the SSA path does not model). The two + * solvers are held byte-identical by the equivalence fuzz (status, bindings, + * sorted facts, def/use telemetry), so the dispatch is a pure performance + * heuristic; the dense solver doubles as that differential oracle. */ export function computeReachingDefs(cfg: FunctionCfg, limits?: ReachingDefsLimits): FunctionDefUse { + // #2201: production auto-selects the solver per function (see + // {@link computeInSetsAuto}) — the SSA solver where it pays off (looping + // functions large enough to amortize construction, incl. the deep nests the + // dense ceiling used to truncate), the dense worklist everywhere else (small + // or loop-free functions, where it is faster). Both are held byte-identical + // by the equivalence fuzz, so the choice is a pure performance heuristic. + return solveReachingDefs(cfg, limits, computeInSetsAuto); +} + +/** + * Dense GEN/KILL monotone worklist — the original (#2082 M2) reaching-defs + * solver. As of #2201 it plays two roles: (1) the production dispatcher + * ({@link computeInSetsAuto}) routes small / loop-free functions, and the + * throw-edge / unreachable-block functions the SSA path does not model, to this + * dense solver; (2) it is the differential equivalence ORACLE the fuzz checks + * the SSA path against. Keep it behavior-frozen — it is the ground truth. + * + * @internal exported for the equivalence fuzz harness (direct dense-vs-sparse + * comparison); the bench drives the production {@link computeReachingDefs}. + */ +export function computeReachingDefsDense( + cfg: FunctionCfg, + limits?: ReachingDefsLimits, +): FunctionDefUse { + return solveReachingDefs(cfg, limits, computeInSetsDense); +} + +/** + * SSA-sparse reaching-defs (#2201) — exposed directly so the equivalence fuzz + * can drive the SSA solver on every eligible CFG (bypassing the production + * size/loop dispatch heuristic in {@link computeInSetsAuto}) and assert + * byte-identity against the dense oracle. See {@link computeInSetsSparse} for + * the algorithm and byte-identical contract. + * + * @internal exported only for the equivalence fuzz harness. + */ +export function computeReachingDefsSparse( + cfg: FunctionCfg, + limits?: ReachingDefsLimits, +): FunctionDefUse { + return solveReachingDefs(cfg, limits, computeInSetsSparse); +} + +/** + * Shared orchestrator: the no-facts / overflow guards, the harvest, the + * adjacency build, the swappable IN-set computation, and the statement sweep. + * Only `computeInSets` differs between the production (sparse) and oracle + * (dense) paths — everything else is identical, which is what makes the two + * byte-identical by construction. + */ +function solveReachingDefs( + cfg: FunctionCfg, + limits: ReachingDefsLimits | undefined, + computeInSets: InSetsComputer, +): FunctionDefUse { if (!cfg.bindings) { return { status: 'no-facts', bindings: [], facts: [], defCount: 0, useCount: 0 }; } @@ -146,51 +306,46 @@ export function computeReachingDefs(cfg: FunctionCfg, limits?: ReachingDefsLimit } } - // ── adjacency (sorted for deterministic merges) ───────────────────────── - // A `throw` edge contributes IN(from) ∪ allDefs(from) to its handler, not - // OUT: an exception can fire BEFORE the block's defs complete (the seed def - // in `let x = seed(); try { x = risky(); } catch { sink(x) }` must reach the - // sink) AND between any two defs of a multi-def coalesced block (the parse - // def in `x = parse(a); x = normalize(x);` is live exactly when normalize - // throws — OUT's last-def-wins misses it). Sound over-approximation; - // monotone, so the fixpoint absorbs it. See mergePreds. - const preds: { from: number; viaThrow: boolean }[][] = Array.from({ length: n }, () => []); - const succs: number[][] = Array.from({ length: n }, () => []); - // Handlers whose IN depends on this block's IN (throw edges) — requeued on - // IN change, since a genned binding can absorb IN growth without changing - // OUT, which would otherwise leave the handler stale. - const throwSuccs: number[][] = Array.from({ length: n }, () => []); - for (const e of cfg.edges) { - // Optional-chained pushes drop out-of-range endpoints defensively — the - // emit path validates via isEmitSafeCfg, but this pure function also runs - // on hand-built CFGs. - succs[e.from]?.push(e.to); - preds[e.to]?.push({ from: e.from, viaThrow: e.kind === 'throw' }); - if (e.kind === 'throw') throwSuccs[e.from]?.push(e.to); + const h = harvestStatementFacts(blocks, n); + const adj = buildAdjacency(cfg, n); + const solved = computeInSets(cfg, n, h, adj, limits); + if (!solved.converged) { + // Did NOT converge within the budget — the in-sets are not at the fixpoint, + // so any facts would be unsound. Bail to a sound empty `truncated` result + // (a coverage gap, not an error), carrying the def/use telemetry gathered. + return { + status: 'truncated', + bindings: cfg.bindings, + facts: [], + defCount: h.defCount, + useCount: h.useCount, + }; } - for (const list of preds) { - list.sort((a, b) => a.from - b.from || Number(a.viaThrow) - Number(b.viaThrow)); - // duplicate (from, throw+non-throw) pairs both survive — the throw leg - // adds IN(from); the merge dedups set-wise. - } - for (const list of succs) list.sort((a, b) => a - b); - // ── per-block GEN + def/use telemetry ──────────────────────────────────── - // gen[b]: bindingIdx → { set, kills }. A MUST def resets the accumulated - // set (kill is total); a MAY def (conditionally-evaluated context — see - // StatementFacts.mayDefs) only ADDS: the binding's incoming defs survive, - // so the transfer is out[x] = kills ? set : in[x] ∪ set. - interface GenEntry { - set: DefSet; - kills: boolean; - } + const maxFacts = limits?.maxFacts && limits.maxFacts > 0 ? limits.maxFacts : Infinity; + const { facts, truncated } = sweepFacts(blocks, solved.reachingAt, h.defLine, maxFacts); + + return { + status: truncated ? 'truncated' : 'computed', + bindings: cfg.bindings, + facts, + defCount: h.defCount, + useCount: h.useCount, + }; +} + +/** + * Per-block GEN + def/use telemetry. gen[b]: bindingIdx → { set, kills }. A + * MUST def resets the accumulated set (kill is total); a MAY def (conditionally- + * evaluated context — see StatementFacts.mayDefs) only ADDS: the binding's + * incoming defs survive, so the transfer is out[x] = kills ? set : in[x] ∪ set. + * allDefsGen[b] is what a throw edge delivers to its handler: an exception can + * fire between any two statements, so every intermediate def may be the live one + * at the handler — IN∪OUT alone misses defs overwritten later in the same + * coalesced block. + */ +function harvestStatementFacts(blocks: FunctionCfg['blocks'], n: number): Harvest { const gen: (Map | null)[] = new Array(n).fill(null); - // allDefsGen[b]: bindingIdx → EVERY def-site key in the block (must + may). - // This is what a throw edge delivers to its handler: an exception can fire - // between any two statements, so every intermediate def may be the live one - // at the handler — IN∪OUT alone misses defs overwritten later in the same - // coalesced block (`try { x = parse(a); x = normalize(x); } catch { sink(x) }` - // — parse's value is exactly what sink sees when normalize throws). const allDefsGen: (Lattice | null)[] = new Array(n).fill(null); const defLine = new Map(); // defKey → source line let defCount = 0; @@ -225,31 +380,73 @@ export function computeReachingDefs(cfg: FunctionCfg, limits?: ReachingDefsLimit gen[b.index] = g; allDefsGen[b.index] = all; } + return { gen, allDefsGen, defLine, defCount, useCount }; +} - // ── iteration order: RPO over reachable blocks, then the rest by index ── - // WTO / loop-aware iteration (Bourdoncle 1993) was evaluated as a fix for the - // O(blocks²) deep-loop-nest blow-up and REJECTED: on the dense-loop benchmark a - // faithful weak-topological-order solver was 104/104 byte-identical to this RPO - // worklist but 0% faster. The cost is inherent to dense-set propagation + - // lattice merges on the iterated dominance frontier, not to visitation order, so - // re-ordering passes buys nothing; the "skip re-evaluating a loop body once its - // header stabilises" shortcut is additionally unsound on irreducible (goto) - // CFGs. The sound, shipped backstop is the maxBlockVisits ceiling below (a - // blocks×64 budget — see emit.ts DEFAULT_PDG_MAX_REACHING_DEF_BLOCK_REVISITS), - // which truncates the pathological nest to a sound-empty result. The only real - // asymptotic fix is SSA-sparse reaching-defs (propagate along def-use chains, not - // dense block sets) — deferred to a tracked follow-up, not a reordering tweak. - const order = reversePostOrder(cfg.entryIndex, succs, n); +/** + * Throw-aware predecessor/successor adjacency, sorted for deterministic merges. + * A `throw` edge contributes IN(from) ∪ allDefs(from) to its handler, not OUT: + * an exception may fire BEFORE the block's defs complete (the seed def in + * `let x = seed(); try { x = risky(); } catch { sink(x) }` must reach the sink) + * AND between any two defs of a multi-def coalesced block. Sound over- + * approximation; monotone, so the fixpoint absorbs it. See mergePreds. + */ +function buildAdjacency(cfg: FunctionCfg, n: number): Adjacency { + const preds: { from: number; viaThrow: boolean }[][] = Array.from({ length: n }, () => []); + const succs: number[][] = Array.from({ length: n }, () => []); + // Handlers whose IN depends on this block's IN (throw edges) — requeued on + // IN change, since a genned binding can absorb IN growth without changing + // OUT, which would otherwise leave the handler stale. + const throwSuccs: number[][] = Array.from({ length: n }, () => []); + for (const e of cfg.edges) { + // Optional-chained pushes drop out-of-range endpoints defensively — the + // emit path validates via isEmitSafeCfg, but this pure function also runs + // on hand-built CFGs. + succs[e.from]?.push(e.to); + preds[e.to]?.push({ from: e.from, viaThrow: e.kind === 'throw' }); + if (e.kind === 'throw') throwSuccs[e.from]?.push(e.to); + } + for (const list of preds) { + list.sort((a, b) => a.from - b.from || Number(a.viaThrow) - Number(b.viaThrow)); + // duplicate (from, throw+non-throw) pairs both survive — the throw leg + // adds IN(from); the merge dedups set-wise. + } + for (const list of succs) list.sort((a, b) => a - b); + return { preds, succs, throwSuccs }; +} + +/** + * DENSE IN-set computer — the original monotone GEN/KILL worklist. Iterates in + * reverse post-order, seeded with every block (unreachable blocks keep ⊥ IN — + * correct, their defs reach nothing). Convergence: sets grow monotonically + * within the finite def-site universe ⇒ ≤ loop-depth+1 passes in practice. + * + * WTO / loop-aware iteration (Bourdoncle 1993) was evaluated as a fix for the + * O(blocks²) deep-loop-nest blow-up and REJECTED (#2195): on the dense-loop + * benchmark a faithful weak-topological-order solver was 104/104 byte-identical + * but 0% faster — the cost is inherent to dense-set propagation + lattice + * merges, not visitation order. The asymptotic fix shipped in #2201: the + * SSA-sparse solver ({@link computeInSetsSparse}). This dense version is retained + * only as the differential equivalence oracle the fuzz checks SSA against. + * + * @internal + */ +function computeInSetsDense( + cfg: FunctionCfg, + n: number, + h: Harvest, + adj: Adjacency, + limits: ReachingDefsLimits | undefined, +): InSetsResult { + const { gen, allDefsGen } = h; + const { preds, succs, throwSuccs } = adj; + const { order } = reversePostOrder(cfg.entryIndex, succs, n); - // ── fixpoint ──────────────────────────────────────────────────────────── const inSets: Lattice[] = new Array(n).fill(EMPTY_LATTICE); const outSets: Lattice[] = new Array(n).fill(EMPTY_LATTICE); const inWorklist = new Array(n).fill(true); let pending = n; - // Fixpoint-iteration ceiling (see ReachingDefsLimits.maxBlockVisits): bound the - // total block dequeues so a pathologically deep loop nest can't drive the - // worklist to O(blocks²). undefined/0 ⇒ unlimited. const maxBlockVisits = limits?.maxBlockVisits && limits.maxBlockVisits > 0 ? limits.maxBlockVisits : Infinity; let blockVisits = 0; @@ -258,13 +455,7 @@ export function computeReachingDefs(cfg: FunctionCfg, limits?: ReachingDefsLimit if (!inWorklist[b]) continue; inWorklist[b] = false; pending -= 1; - if (++blockVisits > maxBlockVisits) { - // Did NOT converge within the budget — the in/out sets are not at the - // fixpoint, so any facts would be unsound. Bail to a sound empty - // `truncated` result (a coverage gap, not an error), carrying the def/use - // telemetry already gathered. - return { status: 'truncated', bindings: cfg.bindings, facts: [], defCount, useCount }; - } + if (++blockVisits > maxBlockVisits) return { converged: false }; const p = preds[b]; const inB: Lattice = @@ -309,17 +500,367 @@ export function computeReachingDefs(cfg: FunctionCfg, limits?: ReachingDefsLimit } } - // ── statement sweep: recover statement-granular def→use facts ─────────── - const maxFacts = limits?.maxFacts && limits.maxFacts > 0 ? limits.maxFacts : Infinity; + return { converged: true, reachingAt: (blockIndex, binding) => inSets[blockIndex]?.get(binding) }; +} + +/** + * SPARSE IN-set computer (#2201) — the production solver. Instead of the dense + * GEN/KILL worklist's per-block lattice fixpoint, it builds pruned SSA for the + * function (Cooper-Harvey-Kennedy dominators → Cytron dominance frontiers and + * φ-placement → stack-based renaming) and answers block-entry reaching-def + * queries by walking the SSA def-use graph. φ-nodes statically capture loop + * merges, so a use's reaching set is recovered without iterating the loop + * (depth-independent), and pass-through blocks carry the dominating definition + * via the rename stack rather than re-materializing a dense lattice at every + * block — the two effects that make it faster than the dense solver on the + * deep-nest and dense-bindings pathologies. + * + * BYTE-IDENTICAL CONTRACT: it computes the same may-reaching-definition SET at + * each block entry as {@link computeInSetsDense}. Order does not matter — {@link + * sweepFacts} sorts each use's reaching keys before the maxFacts cutoff (#2201 + * KTD6) — so only set CONTENTS must match; the equivalence fuzz holds the line. + * + * SCOPE (KTD4): the SSA path covers fully-reachable CFGs with kill/may-def + * transfers, reducible AND irreducible (CHK + Cytron are correct on irreducible + * graphs). It does NOT model throw edges' IN∪allDefs handler semantics or + * propagation among unreachable blocks; functions with either are routed to the + * dense oracle — byte-identical and correct, just not asymptotically faster. + * These are not the perf pathologies (deep nests / dense-bindings are + * throw-free and fully reachable), so the win lands where it matters. + * + * No fixpoint iteration ⇒ the solve always converges in O(program); the + * `maxBlockVisits` ceiling that fired on the dense worklist's deep nests never + * fires here (#2201 acceptance). The bound is honored only on the dense + * fallback path. + * + * @internal + */ +function computeInSetsSparse( + cfg: FunctionCfg, + n: number, + h: Harvest, + adj: Adjacency, + limits: ReachingDefsLimits | undefined, +): InSetsResult { + const nBindings = cfg.bindings?.length ?? 0; + if (nBindings === 0) return { converged: true, reachingAt: () => undefined }; + + const { gen } = h; + const { preds, succs, throwSuccs } = adj; + const entry = cfg.entryIndex; + + // Gate to the dense oracle for the shapes the SSA path does not model. + for (const list of throwSuccs) if (list.length) return computeInSetsDense(cfg, n, h, adj, limits); + // Malformed-input guard: an out-of-range binding index (negative or + // ≥ nBindings — a corrupted/stale durable parsedfile store) would crash the + // SSA path's nBindings-sized arrays (defBlocks[v]/stacks[u]). The dense solver + // tolerates any index (its lattice is a Map), so fall back — keeping the two + // byte-identical AND preserving the graceful per-function degradation the + // dense path gave (a throw here would escape the unguarded taint/harvest call + // sites and lose the whole file's taint layer). See hasEmitSafeFacts (emit.ts). + for (const b of cfg.blocks) { + const stmts = b.statements; + if (!stmts) continue; + for (const s of stmts) { + for (const d of s.defs) + if (d < 0 || d >= nBindings) return computeInSetsDense(cfg, n, h, adj, limits); + for (const u of s.uses) + if (u < 0 || u >= nBindings) return computeInSetsDense(cfg, n, h, adj, limits); + if (s.mayDefs) + for (const d of s.mayDefs) + if (d < 0 || d >= nBindings) return computeInSetsDense(cfg, n, h, adj, limits); + } + } + // Synthetic pre-entry block (#2201): textbook SSA construction assumes the + // entry has no predecessors. A loop back-edge into the entry — or a self-loop + // on it — makes the entry a merge that needs a φ, and the dominance-frontier + // walk degenerates when idom[entry] === entry (it never lands the entry in its + // own frontier). A virtual start node S → entry (S itself has no preds) + // restores the invariant: idom[entry] = S, the entry joins {start ⊔ + // back-edges}, and the implicit start operand contributes ⊥ (an empty rename + // stack). S carries no statements, gen, or uses and is never queried. + const S = n; + const nx = n + 1; + const succsX: number[][] = new Array(nx); + for (let b = 0; b < n; b++) succsX[b] = succs[b] as number[]; + succsX[S] = [entry]; + const dPredsX: number[][] = new Array(nx); + for (let b = 0; b < n; b++) { + // preds[b] is pre-sorted by `from` (buildAdjacency), so duplicate `from` + // values (a throw + non-throw edge to the same handler, or parallel edges) + // are ADJACENT — dedup by skipping consecutive equals instead of a per-block + // Set + spread + sort (#2201 review R9). S = n exceeds every block index, so + // appending it for the entry keeps the list ascending without a re-sort. + const list: number[] = []; + let last = -1; + for (const p of preds[b]) { + if (p.from !== last) { + list.push(p.from); + last = p.from; + } + } + if (b === entry) list.push(S); + dPredsX[b] = list; + } + dPredsX[S] = []; + + // ── dominators (Cooper-Harvey-Kennedy; correct on irreducible CFGs) ── + // RPO rooted at the synthetic entry. `reachX` is the reachability the DFS + // already computed — reused for the unreachable-block gate below instead of a + // separate BFS (#2201 review R8). Because S→entry is S's only edge, reachX[b] + // (b []); + for (let b = 0; b < n; b++) { + const g = gen[b]; + if (g) for (const v of g.keys()) defBlocks[v].push(b); + } + + // ── value-graph nodes: leaves carry def-site keys; internal nodes (φ / + // may-def union) carry operand node ids. reachingSet(node) = union of all + // leaf keys reachable through operands (computed once, cycle-safe, below). + const nodeKeys: (DefSet | null)[] = []; + const nodeOps: number[][] = []; + const newLeaf = (keys: DefSet): number => ( + nodeKeys.push(keys), + nodeOps.push([]), + nodeKeys.length - 1 + ); + const newInternal = (): number => (nodeKeys.push(null), nodeOps.push([]), nodeKeys.length - 1); + + // ── φ-placement: φ for v at the iterated dominance frontier of v's defs ── + const phiNode: (Map | null)[] = new Array(nx).fill(null); + for (let v = 0; v < nBindings; v++) { + const dB = defBlocks[v]; + if (dB.length === 0) continue; + const placed = new Set(); + const inWork = new Set(dB); + const work = [...dB]; + while (work.length) { + const x = work.pop()!; + for (const y of df[x]) { + if (placed.has(y)) continue; + placed.add(y); + let m = phiNode[y]; + if (!m) phiNode[y] = m = new Map(); + m.set(v, newInternal()); + if (!inWork.has(y)) { + inWork.add(y); + work.push(y); + } + } + } + } + + // ── memory bound (#2201 review R1): cap the value graph, else fall back ── + // After φ-placement, nodeKeys.length == the φ-node count — the term that grows + // superlinearly with the input on the deep-loop / dense-binding pathology. + // Renaming below adds at most ~2 nodes per gen entry (already bounded by the + // def-site universe the STMT_STRIDE overflow guard caps). If the projected + // total would exceed the budget, fall back to the dense oracle here — BEFORE + // paying for renaming + Tarjan SCC on a blown-up graph. Byte-identical (dense + // is the equivalence oracle) and bounded (dense honors maxBlockVisits). Mirrors + // the throw-edge / unreachable / OOB-binding gates at the top of this function. + const nodeBudget = + limits?.maxSsaValueGraphNodes && limits.maxSsaValueGraphNodes > 0 + ? limits.maxSsaValueGraphNodes + : DEFAULT_MAX_SSA_VALUE_GRAPH_NODES; + let projectedRenameNodes = 0; + for (let b = 0; b < n; b++) projectedRenameNodes += (gen[b]?.size ?? 0) * 2; + if (nodeKeys.length + projectedRenameNodes > nodeBudget) { + return computeInSetsDense(cfg, n, h, adj, limits); + } + + // ── renaming (iterative dominator-tree DFS, per-binding value stacks) ── + const domChildren: number[][] = Array.from({ length: nx }, () => []); + for (let b = 0; b < nx; b++) if (b !== S && idom[b] !== -1) domChildren[idom[b]].push(b); + for (const list of domChildren) list.sort((a, b) => a - b); + + const stacks: number[][] = Array.from({ length: nBindings }, () => []); + const entryValue: (Map | null)[] = new Array(nx).fill(null); + + const enterBlock = (b: number): number[] => { + const pushed: number[] = []; + const pm = phiNode[b]; + if (pm) + for (const [v, node] of pm) { + stacks[v].push(node); + pushed.push(v); + } + // record block-entry (IN) value for each binding USED here — after φ push, + // before this block's own gen (the sweep applies intra-block defs itself). + // The synthetic entry S has no block ⇒ no statements/gen/uses. + const stmts = cfg.blocks[b]?.statements; + if (stmts) { + let ev: Map | null = null; + for (const s of stmts) + for (const u of s.uses) { + const st = stacks[u]; + if (st.length) { + if (!ev) ev = new Map(); + ev.set(u, st[st.length - 1]); + } + } + entryValue[b] = ev; + } + // apply block gen ⇒ OUT values that flow to successors + const g = gen[b]; + if (g) + for (const [v, ge] of g) { + const st = stacks[v]; + let node: number; + if (ge.kills) { + node = newLeaf(ge.set); + } else { + node = newInternal(); + if (st.length) nodeOps[node].push(st[st.length - 1]); // prior reaching (may-def keeps it) + nodeOps[node].push(newLeaf(ge.set)); + } + st.push(node); + pushed.push(v); + } + // fill successor φ operands with this block's current OUT for each φ binding + for (const s of succsX[b]) { + const sm = phiNode[s]; + if (!sm) continue; + for (const [v, phi] of sm) { + const st = stacks[v]; + if (st.length) nodeOps[phi].push(st[st.length - 1]); + } + } + return pushed; + }; + + const frames: { b: number; ci: number; pushed: number[] }[] = [ + { b: S, ci: 0, pushed: enterBlock(S) }, + ]; + while (frames.length) { + const f = frames[frames.length - 1]; + const kids = domChildren[f.b]; + if (f.ci < kids.length) { + const c = kids[f.ci++]; + frames.push({ b: c, ci: 0, pushed: enterBlock(c) }); + } else { + for (const v of f.pushed) stacks[v].pop(); + frames.pop(); + } + } + + // ── reaching sets per node via SCC condensation (cycle-safe union) ── + // Tarjan condenses the value graph (operand cycles from loop φs collapse to a + // single SCC); a forward pass over the reverse-topo SCC order unions each + // SCC's reaching set from its operands' (alias fast path for single-source + // SCCs — #2201 review R2). Both stages are pure (reaching-defs-graph.ts). + const { sccOf, sccMembers } = tarjanScc(nodeOps); + const reachByScc = condenseReachingSets(sccMembers, sccOf, nodeKeys, nodeOps); + + return { + converged: true, + reachingAt: (blockIndex, binding) => { + const node = entryValue[blockIndex]?.get(binding); + if (node === undefined) return undefined; + const set = reachByScc[sccOf[node]]; + return set.size ? set : undefined; + }, + }; +} + +/** + * Minimum block count below which SSA construction (dominators + dominance + * frontiers + φ-placement + renaming + SCC) does not amortize over the dense + * worklist's single-pass aliasing. Calibrated empirically (~14-block crossover + * for loop-heavy functions; 16 leaves headroom); the dense-bindings + * `rd_scaling_budget` gate in bench/cfg/baselines.json catches a regression if + * this is mistuned. Paired with a reachable-loop check — loop-free functions + * always take the cheaper dense path regardless of size. + */ +const SSA_MIN_BLOCKS = 16; + +/** + * Default ceiling on the SSA-sparse solver's value-graph node count (#2201 + * review R1). Above this the sparse path falls back to the dense oracle (which + * bounds its own work via `maxBlockVisits`), trading the deep-loop full-facts + * win for bounded memory on pathological inputs. Sized FAR above any real or + * benchmarked function: the suite's densest SSA scenarios (`dense-bindings`, + * `deep-nest`) build well under 10⁴ nodes, while the pathology this guards + * (thousands of blocks × hundreds of bindings) builds 10⁶–10⁷. The + * `dense-bindings` / `deep-nest` `rd_scaling_budget` gates in + * bench/cfg/baselines.json fail if this is set so low it forces those scenarios + * onto the dense path. Overridable per-call via + * {@link ReachingDefsLimits.maxSsaValueGraphNodes}. + */ +const DEFAULT_MAX_SSA_VALUE_GRAPH_NODES = 1_000_000; + +/** + * Production solver dispatcher (#2201). The SSA solver beats the dense worklist + * only when there is enough work to amortize SSA construction — a loop (so the + * dense fixpoint pays the loop-depth pass multiplier, or truncates at the + * ceiling) AND a non-trivial block count. Small or loop-free functions, which + * dense solves in one or two cheap aliasing passes, stay on the dense path. + * Because the two solvers are byte-identical (held by the equivalence fuzz), + * this is a pure performance heuristic with no effect on results. + * + * @internal + */ +function computeInSetsAuto( + cfg: FunctionCfg, + n: number, + h: Harvest, + adj: Adjacency, + limits: ReachingDefsLimits | undefined, +): InSetsResult { + if (n >= SSA_MIN_BLOCKS && hasReachableLoop(cfg.entryIndex, adj.succs, n)) { + return computeInSetsSparse(cfg, n, h, adj, limits); + } + return computeInSetsDense(cfg, n, h, adj, limits); +} + +/** + * Statement sweep — recover statement-granular def→use facts from the per-block + * entry reaching lattices, sort them, and apply the maxFacts truncation. SHARED + * by both solvers, and the maxFacts cutoff is where their (intentionally + * different) reaching-set INSERTION orders would otherwise leak into the output: + * the dense worklist seeds keys in RPO fixpoint order, the SSA solver in + * renaming/SCC order, so a loop-carried use's reaching set is the same SET in a + * different order. The byte-identity of a TRUNCATED result therefore does NOT + * come from matching insertion orders — it comes from the KTD6 per-use + * `useKeys.sort()` BELOW, which canonicalizes each use's keys by defKey before + * the cutoff. (The full, untruncated fact array is re-sorted at the end, so the + * pre-sort is a no-op there; its whole purpose is the truncated prefix.) Outer + * emission order — block index, then statement index, then use order — is shared + * structurally and needs no canonicalization. + */ +function sweepFacts( + blocks: FunctionCfg['blocks'], + reachingAt: ReachingAt, + defLine: ReadonlyMap, + maxFacts: number, +): { facts: DefUseFact[]; truncated: boolean } { const facts: DefUseFact[] = []; let truncated = false; + // Scratch buffer for one use's reaching def-keys, reused across every use to + // avoid a per-use array allocation (#2201 review R9). Cleared per use; the + // KTD6 sort below operates on it in place. + const useKeys: number[] = []; outer: for (const b of blocks) { const stmts = b.statements; if (!stmts || stmts.length === 0) continue; - // Lazy overlay of IN — entries are replaced (never mutated) on def, so the - // shared sets stay intact. - let reach: Lattice | null = null; + // Sparse intra-block overlay: only the bindings REDEFINED within this block + // so far. A use's reaching set is the overlay's override if present, else + // the block-entry reaching set (reachingAt). This never materializes the + // full block lattice — the dense O(live-vars) per-block copy the sparse + // solver exists to avoid. + const overlay = new Map(); for (let i = 0; i < stmts.length; i++) { const s = stmts[i]; // A use's binding that the SAME statement also defines could be a @@ -330,17 +871,32 @@ export function computeReachingDefs(cfg: FunctionCfg, limits?: ReachingDefsLimit // self-fact on compound assignments is harmless; missing the // assign-and-test def→use (the most common JS idiom) would be a taint // false negative. May-defs join the self-key set the same way. - const sameStmtDefs = - s.defs.length > 0 || s.mayDefs?.length ? new Set([...s.defs, ...(s.mayDefs ?? [])]) : null; + // def/mayDef arrays are tiny (1–3 entries), so a membership scan over them + // is cheaper than the old per-statement `new Set([...defs, ...mayDefs])` + // (#2201 review R9). `hasSelfDefs` short-circuits pure-use statements. + const hasSelfDefs = s.defs.length > 0 || (s.mayDefs?.length ?? 0) > 0; for (const u of s.uses) { - const reaching = (reach ?? inSets[b.index]).get(u); - const selfKey = sameStmtDefs?.has(u) ? defKey(b.index, i) : undefined; + const reaching = overlay.get(u) ?? reachingAt(b.index, u); + const selfKey = + hasSelfDefs && (s.defs.includes(u) || (s.mayDefs?.includes(u) ?? false)) + ? defKey(b.index, i) + : undefined; if (!reaching && selfKey === undefined) continue; - const keys = - selfKey !== undefined && !reaching?.has(selfKey) - ? [...(reaching ?? []), selfKey] - : [...(reaching ?? [])]; - for (const key of keys) { + // Reuse the scratch buffer instead of spreading a fresh array per use. + useKeys.length = 0; + if (reaching) for (const k of reaching) useKeys.push(k); + if (selfKey !== undefined && !reaching?.has(selfKey)) useKeys.push(selfKey); + // Canonical emission order (#2201 KTD6): sort each use's reaching + // def-sites by defKey (= def block, then def stmt) BEFORE the maxFacts + // cutoff. The full (untruncated) fact array is re-sorted identically at + // the end, so this is a no-op there; its purpose is to make the + // TRUNCATED subset schedule-independent — the reaching SET's insertion + // order is fixpoint-evaluation-order-dependent for loop-carried + // bindings (dense RPO vs sparse change-driven seed different keys + // first), so a pre-sort cutoff is what keeps the two solvers' + // truncated results byte-identical. + useKeys.sort((a, b) => a - b); + for (const key of useKeys) { if (facts.length >= maxFacts) { truncated = true; break outer; @@ -356,16 +912,14 @@ export function computeReachingDefs(cfg: FunctionCfg, limits?: ReachingDefsLimit } if (s.mayDefs?.length) { // Gen WITHOUT kill: the conditional def joins the binding's set. - if (!reach) reach = new Map(inSets[b.index]); const key = defKey(b.index, i); for (const d of s.mayDefs) { - const prior = reach.get(d); - reach.set(d, prior ? unionSets(prior, new Set([key])) : new Set([key])); + const prior = overlay.get(d) ?? reachingAt(b.index, d); + overlay.set(d, prior ? unionSets(prior, new Set([key])) : new Set([key])); } } if (s.defs.length > 0) { - if (!reach) reach = new Map(inSets[b.index]); - for (const d of s.defs) reach.set(d, new Set([defKey(b.index, i)])); // kill + gen + for (const d of s.defs) overlay.set(d, new Set([defKey(b.index, i)])); // kill + gen } } } @@ -379,41 +933,7 @@ export function computeReachingDefs(cfg: FunctionCfg, limits?: ReachingDefsLimit a.bindingIdx - b.bindingIdx, ); - return { - status: truncated ? 'truncated' : 'computed', - bindings: cfg.bindings, - facts, - defCount, - useCount, - }; -} - -/** RPO over blocks reachable from `entry`; unreachable blocks appended by index. */ -function reversePostOrder(entry: number, succs: readonly number[][], n: number): number[] { - const visited = new Array(n).fill(false); - const post: number[] = []; - // Iterative DFS with an explicit phase stack (children pushed in reverse so - // they pop in sorted order — determinism). - const stack: { node: number; childIdx: number }[] = [{ node: entry, childIdx: 0 }]; - visited[entry] = true; - while (stack.length) { - const top = stack[stack.length - 1]; - const children = succs[top.node]; - if (top.childIdx < children.length) { - const next = children[top.childIdx]; - top.childIdx += 1; - if (!visited[next]) { - visited[next] = true; - stack.push({ node: next, childIdx: 0 }); - } - } else { - post.push(top.node); - stack.pop(); - } - } - const order = post.reverse(); - for (let b = 0; b < n; b++) if (!visited[b]) order.push(b); - return order; + return { facts, truncated }; } /** @@ -465,32 +985,3 @@ function mergePreds( } return merged; } - -/** Order-stable union of two def-sets (shares `a` when `b` adds nothing). */ -function unionSets(a: DefSet, b: DefSet): DefSet { - let target = a; - let copied = false; - for (const key of b) { - if (!target.has(key)) { - if (!copied) { - target = new Set(a); - copied = true; - } - target.add(key); - } - } - return target; -} - -/** Per-binding equality with a reference fast path (sets only ever grow). */ -function latticeEquals(a: Lattice, b: Lattice): boolean { - if (a === b) return true; - if (a.size !== b.size) return false; - for (const [k, bSet] of b) { - const aSet = a.get(k); - if (aSet === bSet) continue; - if (!aSet || aSet.size !== bSet.size) return false; - for (const v of bSet) if (!aSet.has(v)) return false; - } - return true; -} diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 7910dde60..769125ad5 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -415,6 +415,14 @@ export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] => // outlive the model that produced them — ANY model-content change // ships as a new digest and repopulates the taint edges. taintModelVersion, + // #2201 review R3: reaching-defs solver identity. The SSA-sparse rewrite + // computes full facts for deep-loop functions the dense worklist used to + // truncate to empty, so an existing `--pdg` index carries stale-truncated + // REACHING_DEF rows. Absent on any pre-#2201 stamp → the key-union + // pdgModeMismatch trips on the first upgraded run and forces the full + // writeback that recomputes the fuller coverage (no `--force` needed). + // Bump this tag on any future change to which facts the solver emits. + reachingDefSolver: 'ssa-sparse-v1', } : undefined; diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index b03db9fc5..9e39a6cc0 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -194,6 +194,19 @@ export interface RepoMeta { * without `--force`. Optional: absent on pre-M3 stamps. */ taintModelVersion?: string; + /** + * Identity of the reaching-definitions solver the persisted REACHING_DEF + * rows were produced under (#2201 review R3). The SSA-sparse rewrite computes + * FULL facts for deep-loop functions the old dense worklist truncated to + * empty (the blocks×64 ceiling no longer fires) — but an existing `--pdg` + * index built under the old solver carries those truncated rows. ABSENT on + * any pre-#2201 stamp, so that absence trips `pdgModeMismatch` on the first + * upgraded run and forces the full writeback that recomputes the now-fuller + * REACHING_DEF coverage without `--force`. Bump the tag on any future change + * that alters which facts the solver emits. Optional for that upgrade reason; + * resolved (always present) on every post-#2201 write. + */ + reachingDefSolver?: string; }; } diff --git a/gitnexus/test/unit/cfg/reaching-defs-equivalence.test.ts b/gitnexus/test/unit/cfg/reaching-defs-equivalence.test.ts new file mode 100644 index 000000000..e56077cba --- /dev/null +++ b/gitnexus/test/unit/cfg/reaching-defs-equivalence.test.ts @@ -0,0 +1,613 @@ +/** + * #2201 — differential equivalence harness for the reaching-defs solvers. + * + * The SSA-sparse rewrite must be BYTE-IDENTICAL to the retained dense GEN/KILL + * oracle ({@link computeReachingDefsDense}). This file is the permanent gate: + * a seeded random-CFG generator drives both solvers and a structural comparator + * asserts identical status / bindings / sorted facts / def-use telemetry. + * + * In U1 both sides run the dense oracle (self-equivalence + corpus-coverage + * sanity); U5 flips the second solver to {@link computeReachingDefs} (sparse) + * — the single change that turns this into the real equivalence gate. + * + * The corpus deliberately covers the shapes where a may-reaching-defs rewrite + * is most likely to diverge: loops + irreducible (goto) topology, throw edges + * (IN∪allDefs handler semantics), may-defs (gen-without-kill), shadowed + * bindings, unreachable blocks, multi-predecessor joins, and the maxFacts / + * maxBlockVisits truncation postures (KTD6 — the truncated SUBSET depends on + * pre-sort emission order, so it must match too). + * + * Default corpus is CI-fast; GITNEXUS_RD_FUZZ_N raises it (the ≥1M run the + * plan calls for) for a deep local/CI-shard pass. + */ +import { describe, it, expect } from 'vitest'; +import { + computeReachingDefs, + computeReachingDefsDense, + computeReachingDefsSparse, + type FunctionDefUse, + type ReachingDefsLimits, +} from '../../../src/core/ingestion/cfg/reaching-defs.js'; +import type { + BindingEntry, + BasicBlockData, + CfgEdgeData, + CfgEdgeKind, + FunctionCfg, + StatementFacts, +} from '../../../src/core/ingestion/cfg/types.js'; + +type Solver = (cfg: FunctionCfg, limits?: ReachingDefsLimits) => FunctionDefUse; + +// ── deterministic PRNG (mulberry32) ─────────────────────────────────────── +function mulberry32(seed: number): () => number { + let a = seed >>> 0; + return () => { + a |= 0; + a = (a + 0x6d2b79f5) | 0; + let t = Math.imul(a ^ (a >>> 15), 1 | a); + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t; + return ((t ^ (t >>> 14)) >>> 0) / 4294967296; + }; +} + +const NON_THROW_KINDS: CfgEdgeKind[] = [ + 'seq', + 'cond-true', + 'cond-false', + 'loop-back', + 'break', + 'continue', + 'return', + 'switch-case', + 'fallthrough', +]; + +// ── random CFG generator ─────────────────────────────────────────────────── +interface GenOpts { + maxBlocks: number; + maxBindings: number; + maxStmtsPerBlock: number; + pNoBindings: number; // chance the whole CFG has bindings:undefined (→ no-facts) + pThrowEdge: number; + pMayDef: number; + pExtraEdge: number; // per-block chance of an extra random edge + pShadowName: number; +} + +const DEFAULT_GEN: GenOpts = { + // Span both sides of the production SSA dispatch threshold (SSA_MIN_BLOCKS=16): + // CFGs below it route the auto-dispatcher (computeReachingDefs) to dense, those + // above with a reachable loop route it to the SSA path — so the corpus + // differentially exercises BOTH branches of computeInSetsAuto, not just the + // forced-SSA computeReachingDefsSparse entry. See the hadLargeLoop coverage + // guard below. + maxBlocks: 36, + maxBindings: 8, + maxStmtsPerBlock: 4, + pNoBindings: 0.03, + pThrowEdge: 0.12, + pMayDef: 0.18, + pExtraEdge: 0.9, + pShadowName: 0.4, +}; + +function genCfg(seed: number, opts: GenOpts = DEFAULT_GEN): FunctionCfg { + const rnd = mulberry32(seed); + const int = (n: number) => Math.floor(rnd() * n); + const n = 1 + int(opts.maxBlocks); // ≥1 block (entry) + + // bindings — small name pool so shadowing collisions happen; distinct + // declLine/declColumn keep non-synthetic bindings' keys distinct. + const noBindings = rnd() < opts.pNoBindings; + const nBindings = noBindings ? 0 : int(opts.maxBindings + 1); + const namePool = ['a', 'b', 'c', 'd', 'e']; + const kinds: BindingEntry['kind'][] = ['var', 'let', 'const', 'param', 'catch']; + const bindings: BindingEntry[] = []; + for (let i = 0; i < nBindings; i++) { + const shadow = rnd() < opts.pShadowName; + bindings.push({ + name: shadow ? namePool[int(namePool.length)] : `v${i}`, + declLine: 100 + i, + declColumn: i, + kind: kinds[int(kinds.length)], + ...(rnd() < 0.08 ? { synthetic: true } : {}), + }); + } + + const pickBindings = (max: number): number[] => { + if (nBindings === 0) return []; + const out: number[] = []; + const count = int(max + 1); + for (let k = 0; k < count; k++) out.push(int(nBindings)); + return out; + }; + + // blocks (block 0 = entry; some blocks get no statements like synthetic + // ENTRY/EXIT to exercise the skip paths). + const blocks: BasicBlockData[] = []; + for (let b = 0; b < n; b++) { + const stmtCount = b === 0 && rnd() < 0.5 ? int(2) : int(opts.maxStmtsPerBlock + 1); + const statements: StatementFacts[] = []; + for (let i = 0; i < stmtCount; i++) { + const defs = pickBindings(2); + const uses = pickBindings(3); + const mayDefs = rnd() < opts.pMayDef ? pickBindings(1) : []; + statements.push({ + line: b * 100 + i + 1, + defs, + uses, + ...(mayDefs.length ? { mayDefs } : {}), + }); + } + blocks.push({ + index: b, + startLine: b * 100, + endLine: b * 100 + stmtCount, + text: `B${b}`, + kind: b === 0 ? 'entry' : b === n - 1 ? 'exit' : 'normal', + // bindings:undefined ⇒ no-facts: drop statements entirely so it mirrors a + // pre-M2 CFG (the solver keys no-facts off cfg.bindings, but a realistic + // no-facts CFG also lacks statements). + ...(noBindings ? {} : { statements }), + }); + } + + // edges — a probabilistic spine (entry chain) for reachability + random + // extra edges that produce loops, irreducible topology, and unreachable + // blocks. Throw edges target a random handler block. + const edges: CfgEdgeData[] = []; + const addEdge = (from: number, to: number, kind: CfgEdgeKind) => { + if (from >= 0 && from < n && to >= 0 && to < n) edges.push({ from, to, kind }); + }; + for (let b = 0; b < n - 1; b++) { + if (rnd() < 0.75) addEdge(b, b + 1, 'seq'); + } + for (let b = 0; b < n; b++) { + if (rnd() < opts.pExtraEdge) { + const to = int(n); // any target → forward / back / self / cross edges + const throwIt = rnd() < opts.pThrowEdge; + addEdge(b, to, throwIt ? 'throw' : NON_THROW_KINDS[int(NON_THROW_KINDS.length)]); + } + } + + return { + filePath: 'fuzz.ts', + functionStartLine: 1, + functionEndLine: n * 100, + functionStartColumn: 0, + entryIndex: 0, + exitIndex: n - 1, + blocks, + edges, + ...(noBindings ? {} : { bindings }), + }; +} + +// ── hand-built canonical hard CFGs (guaranteed shape coverage) ───────────── +// These pin the gnarly shapes the random generator hits only probabilistically. +function canonicalHardCfgs(): FunctionCfg[] { + const mk = ( + blocks: BasicBlockData[], + edges: CfgEdgeData[], + bindings: BindingEntry[], + ): FunctionCfg => ({ + filePath: 'canon.ts', + functionStartLine: 1, + functionEndLine: 999, + functionStartColumn: 0, + entryIndex: 0, + exitIndex: blocks.length - 1, + blocks, + edges, + bindings, + }); + const bind = (name: string, line: number): BindingEntry => ({ + name, + declLine: line, + declColumn: 0, + kind: 'let', + }); + const blk = (index: number, statements: StatementFacts[]): BasicBlockData => ({ + index, + startLine: index * 10, + endLine: index * 10 + statements.length, + text: `B${index}`, + kind: index === 0 ? 'entry' : 'normal', + statements, + }); + const st = ( + line: number, + defs: number[], + uses: number[], + mayDefs?: number[], + ): StatementFacts => ({ + line, + defs, + uses, + ...(mayDefs ? { mayDefs } : {}), + }); + + const out: FunctionCfg[] = []; + + // (1) Irreducible two-entry loop: 0→1, 0→2, 1→2, 2→1. binding x def in 1, use in 2 & 1. + out.push( + mk( + [blk(0, [st(1, [0], [])]), blk(1, [st(2, [0], [0])]), blk(2, [st(3, [], [0])])], + [ + { from: 0, to: 1, kind: 'cond-true' }, + { from: 0, to: 2, kind: 'cond-false' }, + { from: 1, to: 2, kind: 'seq' }, + { from: 2, to: 1, kind: 'loop-back' }, + ], + [bind('x', 1)], + ), + ); + + // (2) Self-loop with may-def: block 1 loops to itself; x may-def + use. + out.push( + mk( + [blk(0, [st(1, [0], [])]), blk(1, [st(2, [], [0], [0])])], + [ + { from: 0, to: 1, kind: 'seq' }, + { from: 1, to: 1, kind: 'loop-back' }, + ], + [bind('x', 1)], + ), + ); + + // (3) try/catch throw edge: 0 (x=1), 1 (x=parse; x=normalize) -throw-> 2 (use x). + out.push( + mk( + [ + blk(0, [st(1, [0], [])]), + blk(1, [st(2, [0], []), st(3, [0], [0])]), + blk(2, [st(4, [], [0])]), + ], + [ + { from: 0, to: 1, kind: 'seq' }, + { from: 1, to: 2, kind: 'seq' }, + { from: 1, to: 2, kind: 'throw' }, + ], + [bind('x', 1)], + ), + ); + + // (4) Diamond merge: both arm defs reach the join use. + out.push( + mk( + [ + blk(0, [st(1, [], [])]), + blk(1, [st(2, [0], [])]), + blk(2, [st(3, [0], [])]), + blk(3, [st(4, [], [0])]), + ], + [ + { from: 0, to: 1, kind: 'cond-true' }, + { from: 0, to: 2, kind: 'cond-false' }, + { from: 1, to: 3, kind: 'seq' }, + { from: 2, to: 3, kind: 'seq' }, + ], + [bind('x', 1)], + ), + ); + + // (5) Unreachable block carrying a def (block 2 not reachable from entry). + out.push( + mk( + [blk(0, [st(1, [0], [0])]), blk(1, [st(2, [], [0])]), blk(2, [st(3, [0], [0])])], + [{ from: 0, to: 1, kind: 'seq' }], + [bind('x', 1)], + ), + ); + + // (6) Back-edge into the ENTRY block: 0 (def+use x) → 1 (def+use x) → 0 (loop + // back to entry) and 1 → 2 (exit, use x). The SSA solver's synthetic pre-entry + // node exists precisely for this — the entry is a loop header, so x's loop- + // carried def must reach the entry's own use. Pins that path deterministically. + out.push( + mk( + [blk(0, [st(1, [0], [0])]), blk(1, [st(2, [0], [0])]), blk(2, [st(3, [], [0])])], + [ + { from: 0, to: 1, kind: 'seq' }, + { from: 1, to: 0, kind: 'loop-back' }, + { from: 1, to: 2, kind: 'cond-false' }, + ], + [bind('x', 1)], + ), + ); + + // (7) Malformed input: an OUT-OF-RANGE binding index (≥ nBindings, e.g. from a + // corrupted/stale durable store) in a looping CFG. The dense solver tolerates + // it (its lattice is a Map keyed by index); the SSA path must fall back to + // dense rather than crash its nBindings-sized arrays. Asserting byte-identity + // here pins that gate — without it, the SSA path throws and the differential + // comparison can never reach this divergent input (the generator only ever + // emits in-range indices). + out.push( + mk( + [blk(0, [st(1, [0], [])]), blk(1, [st(2, [3], [3])]), blk(2, [st(3, [], [0])])], + [ + { from: 0, to: 1, kind: 'seq' }, + { from: 1, to: 1, kind: 'loop-back' }, + { from: 1, to: 2, kind: 'cond-false' }, + ], + [bind('x', 1)], // nBindings = 1, so binding index 3 in block 1 is out of range + ), + ); + + return out; +} + +// ── structural comparator ────────────────────────────────────────────────── +function serializeFact(f: FunctionDefUse['facts'][number]): string { + return ( + `${f.def.blockIndex}:${f.def.stmtIndex}@${f.def.line}` + + `->${f.use.blockIndex}:${f.use.stmtIndex}@${f.use.line}#${f.bindingIdx}` + ); +} + +/** Returns null when byte-identical, else a human-readable first divergence. */ +function diffDefUse(a: FunctionDefUse, b: FunctionDefUse): string | null { + if (a.status !== b.status) return `status: ${a.status} vs ${b.status}`; + if (a.defCount !== b.defCount) return `defCount: ${a.defCount} vs ${b.defCount}`; + if (a.useCount !== b.useCount) return `useCount: ${a.useCount} vs ${b.useCount}`; + if (a.bindings.length !== b.bindings.length) { + return `bindings.length: ${a.bindings.length} vs ${b.bindings.length}`; + } + if (a.facts.length !== b.facts.length) { + return `facts.length: ${a.facts.length} vs ${b.facts.length}`; + } + for (let i = 0; i < a.facts.length; i++) { + const fa = serializeFact(a.facts[i]); + const fb = serializeFact(b.facts[i]); + if (fa !== fb) return `fact[${i}]: ${fa} vs ${fb}`; + } + return null; +} + +// ── corpus shape classifier (coverage guard) ─────────────────────────────── +interface ShapeFlags { + hasLoop: boolean; + hasThrow: boolean; + hasMayDef: boolean; + hasShadow: boolean; + hasMultiPred: boolean; + hasUnreachable: boolean; + // ≥16-block CFG (SSA_MIN_BLOCKS) with a loop reachable from entry — the exact + // shape the production dispatcher (computeInSetsAuto) sends to the SSA solver. + // Asserting it proves the auto-dispatcher's SSA branch is differentially fuzzed. + hadLargeLoop: boolean; + hadComputed: boolean; + hadTruncated: boolean; + hadNoFacts: boolean; +} + +function classify(cfg: FunctionCfg, flags: ShapeFlags): void { + const n = cfg.blocks.length; + if (cfg.edges.some((e) => e.kind === 'throw')) flags.hasThrow = true; + if (cfg.blocks.some((b) => b.statements?.some((s) => s.mayDefs?.length))) flags.hasMayDef = true; + if (cfg.bindings) { + const names = cfg.bindings.map((b) => b.name); + if (new Set(names).size < names.length) flags.hasShadow = true; + } + const predCount = new Array(n).fill(0); + for (const e of cfg.edges) if (e.to >= 0 && e.to < n) predCount[e.to]++; + if (predCount.some((c) => c >= 2)) flags.hasMultiPred = true; + + // cycle detection (DFS rec-stack) over the whole graph + const succ: number[][] = Array.from({ length: n }, () => []); + for (const e of cfg.edges) + if (e.from >= 0 && e.from < n && e.to >= 0 && e.to < n) succ[e.from].push(e.to); + const color = new Array(n).fill(0); // 0=white 1=gray 2=black + const hasCycleFrom = (start: number): boolean => { + const stack: { node: number; idx: number }[] = [{ node: start, idx: 0 }]; + color[start] = 1; + while (stack.length) { + const top = stack[stack.length - 1]; + if (top.idx < succ[top.node].length) { + const nx = succ[top.node][top.idx++]; + if (color[nx] === 1) return true; + if (color[nx] === 0) { + color[nx] = 1; + stack.push({ node: nx, idx: 0 }); + } + } else { + color[top.node] = 2; + stack.pop(); + } + } + return false; + }; + for (let s = 0; s < n; s++) if (color[s] === 0 && hasCycleFrom(s)) flags.hasLoop = true; + + // reachability from entry + const seen = new Array(n).fill(false); + const q = [cfg.entryIndex]; + seen[cfg.entryIndex] = true; + while (q.length) { + const x = q.pop()!; + for (const y of succ[x]) if (!seen[y]) ((seen[y] = true), q.push(y)); + } + if (seen.some((v, i) => !v && i < n)) flags.hasUnreachable = true; + + // loop reachable from entry (matches the dispatcher's hasReachableLoop) + + // ≥16 blocks ⇒ the production auto-dispatcher routes this CFG to the SSA path. + const c2 = new Array(n).fill(0); + let entryLoop = false; + const st2: { node: number; idx: number }[] = [{ node: cfg.entryIndex, idx: 0 }]; + c2[cfg.entryIndex] = 1; + while (st2.length && !entryLoop) { + const top = st2[st2.length - 1]; + if (top.idx < succ[top.node].length) { + const v = succ[top.node][top.idx++]; + if (c2[v] === 1) entryLoop = true; + else if (c2[v] === 0) ((c2[v] = 1), st2.push({ node: v, idx: 0 })); + } else ((c2[top.node] = 2), st2.pop()); + } + if (n >= 16 && entryLoop) flags.hadLargeLoop = true; +} + +// ── corpus runner ────────────────────────────────────────────────────────── +interface CorpusResult { + checked: number; + flags: ShapeFlags; + firstFailure: string | null; +} + +function runCorpus( + left: Solver, + right: Solver, + count: number, + baseSeed: number, + // maxBlockVisits has DIFFERENT (intentional) semantics across the solvers: the + // dense worklist counts block dequeues against it; the SSA solver has no + // fixpoint iteration and ignores it in its main path (it only flows through to + // the dense fallback for throw-edge/unreachable functions). So a tight budget + // truncates them at different points. Perturb it only when comparing a solver + // against ITSELF (same semantics); cross-solver byte-identity is asserted with + // the budget unlimited (both fully converge). + perturbBlockVisits = true, +): CorpusResult { + const flags: ShapeFlags = { + hasLoop: false, + hasThrow: false, + hasMayDef: false, + hasShadow: false, + hadLargeLoop: false, + hasMultiPred: false, + hasUnreachable: false, + hadComputed: false, + hadTruncated: false, + hadNoFacts: false, + }; + let firstFailure: string | null = null; + let checked = 0; + + const check = (cfg: FunctionCfg, limits: ReachingDefsLimits | undefined, label: string): void => { + const a = left(cfg, limits); + const b = right(cfg, limits); + const d = diffDefUse(a, b); + checked++; + if (a.status === 'computed') flags.hadComputed = true; + if (a.status === 'truncated') flags.hadTruncated = true; + if (a.status === 'no-facts') flags.hadNoFacts = true; + if (d && !firstFailure) firstFailure = `${label}: ${d}`; + }; + + // canonical hard CFGs first (under several limit postures) + for (const [i, cfg] of canonicalHardCfgs().entries()) { + classify(cfg, flags); + check(cfg, undefined, `canon[${i}]`); + check(cfg, { maxFacts: 1 }, `canon[${i}]/maxFacts=1`); + check(cfg, { maxFacts: 2 }, `canon[${i}]/maxFacts=2`); + if (perturbBlockVisits) check(cfg, { maxBlockVisits: 2 }, `canon[${i}]/maxBlockVisits=2`); + } + + // random corpus + for (let i = 0; i < count; i++) { + const seed = baseSeed + i; + const cfg = genCfg(seed); + classify(cfg, flags); + check(cfg, undefined, `seed=${seed}`); + // exercise truncation on ~1/4 of cases (small maxFacts) and the block-visit + // ceiling on ~1/8 — both must match byte-for-byte (KTD6). + if (i % 4 === 0) check(cfg, { maxFacts: 1 + (i % 3) }, `seed=${seed}/maxFacts`); + if (perturbBlockVisits && i % 8 === 0) { + check(cfg, { maxBlockVisits: 1 + (i % 4) }, `seed=${seed}/maxBlockVisits`); + } + } + + return { checked, flags, firstFailure }; +} + +const CORPUS_N = Number(process.env.GITNEXUS_RD_FUZZ_N ?? 1500); + +describe('#2201 reaching-defs differential equivalence', () => { + it('dense oracle is self-consistent and the comparator + generator are sound', () => { + // U1 baseline: dense-vs-dense MUST be byte-identical (proves the harness). + const r = runCorpus(computeReachingDefsDense, computeReachingDefsDense, CORPUS_N, 0x2201); + expect(r.firstFailure).toBeNull(); + expect(r.checked).toBeGreaterThan(CORPUS_N); + }); + + it('the corpus exercises every divergence-prone shape (coverage guard)', () => { + const r = runCorpus(computeReachingDefsDense, computeReachingDefsDense, CORPUS_N, 0x2201); + const f = r.flags; + expect(f.hasLoop, 'loops').toBe(true); + expect(f.hasThrow, 'throw edges').toBe(true); + expect(f.hasMayDef, 'may-defs').toBe(true); + expect(f.hasShadow, 'shadowed bindings').toBe(true); + expect(f.hasMultiPred, 'multi-pred joins').toBe(true); + expect(f.hasUnreachable, 'unreachable blocks').toBe(true); + expect(f.hadLargeLoop, '≥16-block looping CFGs (production SSA dispatch path)').toBe(true); + expect(f.hadComputed, 'computed results').toBe(true); + expect(f.hadTruncated, 'truncated results').toBe(true); + expect(f.hadNoFacts, 'no-facts results').toBe(true); + }); + + it('is deterministic — a fixed seed yields a byte-identical corpus across runs', () => { + const a = runCorpus(computeReachingDefsDense, computeReachingDefsDense, 200, 0xfeed); + const b = runCorpus(computeReachingDefsDense, computeReachingDefsDense, 200, 0xfeed); + expect(a.checked).toBe(b.checked); + expect(a.flags).toEqual(b.flags); + }); + + it('the SPARSE solver is byte-identical to the dense oracle (#2201 gate)', () => { + // The load-bearing equivalence gate: sparse vs dense across the full corpus, + // budget unlimited so both fully converge. maxFacts truncation IS compared + // (it must match byte-for-byte — KTD6); maxBlockVisits is not (the two count + // different things on purpose — that contrast is the no-regression test). + const r = runCorpus( + computeReachingDefsSparse, + computeReachingDefsDense, + CORPUS_N, + 0x2201, + /* perturbBlockVisits */ false, + ); + expect(r.firstFailure).toBeNull(); + expect(r.flags.hadComputed && r.flags.hadTruncated).toBe(true); + }); + + it('PRODUCTION computeReachingDefs is byte-identical to the dense oracle', () => { + // U1: computeReachingDefs delegates to dense (trivially green). U5 swaps it + // to the sparse solver — this stays the production-entry gate. + const r = runCorpus( + computeReachingDefs, + computeReachingDefsDense, + CORPUS_N, + 0x5eed, + /* perturbBlockVisits */ false, + ); + expect(r.firstFailure).toBeNull(); + }); + + it('sparse never regresses coverage under the production block-visit budget', () => { + // Production posture: emit passes maxBlockVisits = blocks × 64. The contract + // is one-directional — wherever the dense solver COMPUTES, the sparse solver + // must also compute and produce identical facts (no lost REACHING_DEF + // coverage). The reverse is allowed and desired: sparse may compute deep + // nests the dense solver truncates (the #2201 ceiling-stops-firing win). + let regressions = 0; + let firstRegression: string | null = null; + for (let i = 0; i < CORPUS_N; i++) { + const cfg = genCfg(0xc0de + i); + const budget = { maxBlockVisits: cfg.blocks.length * 64 }; + const dense = computeReachingDefsDense(cfg, budget); + const sparse = computeReachingDefsSparse(cfg, budget); + if (dense.status === 'computed') { + const d = diffDefUse(dense, sparse); + if (d) { + regressions++; + if (!firstRegression) firstRegression = `seed=${0xc0de + i}: ${d}`; + } + } + } + expect(firstRegression).toBeNull(); + expect(regressions).toBe(0); + }); +}); + +// Re-exported for U5 and future harness reuse. +export { genCfg, canonicalHardCfgs, diffDefUse, runCorpus, classify }; +export type { Solver, ShapeFlags, CorpusResult }; diff --git a/gitnexus/test/unit/cfg/reaching-defs.test.ts b/gitnexus/test/unit/cfg/reaching-defs.test.ts index 244bc8f58..9fd8c39f0 100644 --- a/gitnexus/test/unit/cfg/reaching-defs.test.ts +++ b/gitnexus/test/unit/cfg/reaching-defs.test.ts @@ -8,6 +8,8 @@ import { } from '../../../src/core/ingestion/cfg/visitors/typescript.js'; import { computeReachingDefs, + computeReachingDefsDense, + computeReachingDefsSparse, type DefUseFact, } from '../../../src/core/ingestion/cfg/reaching-defs.js'; import type { @@ -371,6 +373,115 @@ describe('computeReachingDefs — determinism and convergence', () => { expect(capped.facts).toEqual([]); expect(capped.defCount).toBe(full.defCount); }); + + it('#2201 R5: the ceiling fires on the dense oracle but not on the SSA solver', () => { + // Contrast the two solvers on a looping CFG under a budget below the dense + // worklist's convergence: the dense oracle truncates to a sound-empty result + // (the ceiling fires), while the SSA solver — which has no fixpoint + // iteration — always converges (the ceiling that fired on the dense worklist + // effectively never fires). The facts the SSA solver computes are identical + // to the dense oracle's unbounded result. This is the #2201 acceptance: the + // blocks×64 ceiling stops firing on deep loops. + const blocks: BlockSpec[] = [{}, {}, { stmts: [stmt(3, [0], [0])] }]; + const edges: [number, number][] = [ + [0, 2], + [2, 2], // self-loop → the dense fixpoint must re-visit block 2 + [2, 1], + ]; + const denseFull = computeReachingDefsDense(mkCfg(blocks, edges, ['x'])); + const denseCeiling = computeReachingDefsDense(mkCfg(blocks, edges, ['x']), { + maxBlockVisits: 1, + }); + const sparse = computeReachingDefsSparse(mkCfg(blocks, edges, ['x']), { maxBlockVisits: 1 }); + + expect(denseFull.status).toBe('computed'); + expect(denseFull.facts.length).toBeGreaterThan(0); + expect(denseCeiling.status).toBe('truncated'); // ceiling fires on the dense worklist + expect(sparse.status).toBe('computed'); // SSA ignores the ceiling — it never fires + expect(render(sparse.facts)).toEqual(render(denseFull.facts)); // and the facts match + }); + + it('#2201: an out-of-range binding index in a ≥16-block loop does NOT crash the SSA path', () => { + // A corrupted/stale store can carry a binding index ≥ nBindings. The dense + // solver tolerates it (Map-keyed lattice); the SSA path's nBindings-sized + // arrays would throw. The production dispatcher routes ≥16-block looping + // functions to SSA, so without the malformed-input gate the throw would + // escape the (unguarded) taint/harvest callers and lose a whole file's taint + // layer. The gate falls back to dense — no throw, byte-identical to dense. + const blocks: BlockSpec[] = [{ stmts: [stmt(1, [0], [])] }]; + const edges: [number, number][] = []; + for (let i = 1; i <= 18; i++) { + blocks.push({ stmts: [stmt(i + 1, i === 1 ? [5] : [0], [i === 1 ? 5 : 0])] }); // block 1 uses/defs OOB index 5 + edges.push([i - 1, i]); + } + edges.push([18, 1]); // back-edge → loop; 19 blocks total, ≥16 → SSA dispatch + const cfg = mkCfg(blocks, edges, ['x']); // nBindings = 1; index 5 is out of range + expect(cfg.blocks.length).toBeGreaterThanOrEqual(16); + let prod: ReturnType | undefined; + expect(() => { + prod = computeReachingDefs(cfg); // must NOT throw (gate → dense fallback) + }).not.toThrow(); + const dense = computeReachingDefsDense(cfg); + expect(prod!.status).toBe(dense.status); + expect(render(prod!.facts)).toEqual(render(dense.facts)); // byte-identical to the tolerant dense path + }); + + it('#2201 R1: an oversized SSA value graph falls back to the dense oracle (byte-identical)', () => { + // A ≥16-block looping multi-binding CFG → the production dispatcher routes it + // to the SSA-sparse path. `maxFacts` bounds only fact materialization, not the + // φ/value-graph the sparse path builds first; `maxSsaValueGraphNodes` caps that + // graph and falls back to the dense oracle when it would be too large. Because + // the fallback is byte-identical to dense, the routing flip is made OBSERVABLE + // via a tight `maxBlockVisits`: dense honors the ceiling (truncates), the SSA + // path ignores it (computes) — so the same budget yields different statuses + // depending on which solver ran. + const K = 4; // bindings + const blocks: BlockSpec[] = [{}, {}]; // 0 entry, 1 exit + const edges: [number, number][] = [[0, 2]]; + const BODY = 18; // body blocks 2..19 → 20 blocks total (≥ SSA_MIN_BLOCKS) + for (let i = 0; i < BODY; i++) { + const b = 2 + i; + blocks[b] = { stmts: [stmt(b * 10, [i % K], [(i + 1) % K])] }; + if (i < BODY - 1) edges.push([b, b + 1]); + } + edges.push([2 + BODY - 1, 2]); // back-edge → reachable loop (forces SSA dispatch) + edges.push([2, 1]); // exit + const bindings = Array.from({ length: K }, (_, i) => `v${i}`); + const mk = () => mkCfg(blocks, edges, bindings); + expect(mk().blocks.length).toBeGreaterThanOrEqual(16); + + const denseFull = computeReachingDefsDense(mk()); + expect(denseFull.status).toBe('computed'); + expect(denseFull.facts.length).toBeGreaterThan(0); + + // Tiny node cap, unbounded visits → falls back to dense → byte-identical. + const cappedUnbounded = computeReachingDefs(mk(), { maxSsaValueGraphNodes: 1 }); + expect(cappedUnbounded.status).toBe(denseFull.status); + expect(render(cappedUnbounded.facts)).toEqual(render(denseFull.facts)); + + // Tiny node cap + tight block-visit budget → fallback to dense, whose ceiling + // then fires (truncated, empty). This is the observable proof the cap diverted + // the solve to the dense path. + const cappedBudgeted = computeReachingDefs(mk(), { + maxSsaValueGraphNodes: 1, + maxBlockVisits: 1, + }); + expect(cappedBudgeted.status).toBe('truncated'); + expect(cappedBudgeted.facts).toEqual([]); + + // Default (huge) cap + the SAME tight budget → SSA path runs (no fixpoint + // iteration → ceiling never fires) and computes the full facts. + const uncapped = computeReachingDefs(mk(), { maxBlockVisits: 1 }); + expect(uncapped.status).toBe('computed'); + expect(render(uncapped.facts)).toEqual(render(denseFull.facts)); + + // Boundary monotonicity: a cap well above the graph stays on SSA (computes + // under the tight budget), a cap well below falls back (truncates). + const above = computeReachingDefs(mk(), { maxSsaValueGraphNodes: 100_000, maxBlockVisits: 1 }); + expect(above.status).toBe('computed'); + const below = computeReachingDefs(mk(), { maxSsaValueGraphNodes: 5, maxBlockVisits: 1 }); + expect(below.status).toBe('truncated'); + }); }); describe('computeReachingDefs — parser-direct acceptance (with U1/U2)', () => { diff --git a/gitnexus/test/unit/pdg-mode-flip.test.ts b/gitnexus/test/unit/pdg-mode-flip.test.ts index 5e8dbe2f9..585747eb7 100644 --- a/gitnexus/test/unit/pdg-mode-flip.test.ts +++ b/gitnexus/test/unit/pdg-mode-flip.test.ts @@ -176,6 +176,42 @@ describe('pdgModeMismatch — pre-M5→M5 CDG-cap stamp upgrade (#2085 M5, pure) }); }); +describe('pdgModeMismatch — pre-#2201→SSA reaching-defs solver upgrade (#2201 review R3, pure)', () => { + it('resolvePdgConfig stamps the reaching-defs solver identity', async () => { + const { resolvePdgConfig } = await import('../../src/core/run-analyze.js'); + const stamp = resolvePdgConfig({ pdg: true }); + expect(stamp?.reachingDefSolver).toBe('ssa-sparse-v1'); + }); + + it('a pre-#2201 stamp (no solver key) mismatches the SSA request — upgrade recomputes truncated deep-loop facts', async () => { + const { pdgModeMismatch } = await import('../../src/core/run-analyze.js'); + // What a pre-#2201 (M5-era) run wrote: every cap + model digest, but NO + // reachingDefSolver. The key-union comparator sees 'ssa-sparse-v1' !== + // undefined and trips the full writeback that recomputes the now-fuller + // REACHING_DEF coverage — the deep-loop functions the dense worklist + // truncated to empty at the blocks×64 ceiling now compute full facts. + const m5Stamp = { + maxFunctionLines: 2000, + maxEdgesPerFunction: 5000, + maxReachingDefEdgesPerFunction: 4000, + maxCdgEdgesPerFunction: 5000, + maxTaintFindingsPerFunction: 200, + maxTaintHops: 32, + maxInterprocFindings: 2000, + maxInterprocHops: 32, + maxInterprocEdges: 1000, + taintModelVersion, + }; + expect(pdgModeMismatch(m5Stamp, { pdg: true })).toBe(true); + }); + + it('an identical post-#2201 stamp compares equal (no spurious re-analysis churn)', async () => { + const { pdgModeMismatch, resolvePdgConfig } = await import('../../src/core/run-analyze.js'); + const stamp = resolvePdgConfig({ pdg: true }); + expect(pdgModeMismatch(stamp, { pdg: true })).toBe(false); + }); +}); + describe('detect_changes BasicBlock exclusion (#2082 U7)', () => { it('the symbol-overlap id-prefix filter excludes exactly the BasicBlock rows', async () => { const repo = await setupMiniRepo(); @@ -258,6 +294,7 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxInterprocHops: 32, maxInterprocEdges: 1000, taintModelVersion, + reachingDefSolver: 'ssa-sparse-v1', }); expect(stamped!.incrementalInProgress).toBeUndefined(); // cleared on success @@ -314,6 +351,7 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxInterprocHops: 32, maxInterprocEdges: 1000, taintModelVersion, + reachingDefSolver: 'ssa-sparse-v1', }); // The CFG layer survives a rebuild under a tighter edge cap (blocks are // never capped, only edges). diff --git a/gitnexus/test/unit/run-analyze.test.ts b/gitnexus/test/unit/run-analyze.test.ts index d89114360..b317d6824 100644 --- a/gitnexus/test/unit/run-analyze.test.ts +++ b/gitnexus/test/unit/run-analyze.test.ts @@ -349,6 +349,10 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { // Content digest, not a tunable cap — pinned via the exported constant // (its VALUE changes whenever the built-in model changes, by design). taintModelVersion, + // Solver identity, not a tunable cap — always stamped on a pdg-on run + // (#2201 review R3). Bumps when the reaching-defs solver's emitted facts + // change; absence on a pre-#2201 stamp forces a re-analysis. + reachingDefSolver: 'ssa-sparse-v1', }; it('resolvePdgConfig: pdg-off run resolves to undefined (the meta field is omitted)', async () => { @@ -384,6 +388,7 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { maxInterprocHops: 0, maxInterprocEdges: 0, taintModelVersion, // not a cap — always stamped on a pdg-on run + reachingDefSolver: 'ssa-sparse-v1', // solver identity — always stamped (#2201 R3) }); }); diff --git a/gitnexus/tsconfig.json b/gitnexus/tsconfig.json index 6b82c6d7d..9a3fe9ccd 100644 --- a/gitnexus/tsconfig.json +++ b/gitnexus/tsconfig.json @@ -12,6 +12,7 @@ "resolveJsonModule": true, "forceConsistentCasingInFileNames": true, "declaration": true, + "stripInternal": true, "types": ["node"] }, "include": ["src/**/*"] From df08ecc3979d02b8e9f60c290f8c99e0355aa523 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 15 Jun 2026 19:40:59 +0100 Subject: [PATCH 04/26] perf(lbug): cut graph-DB emit/persistence wall time (#2203) (#2215) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(lbug): add PROF_LBUG_LOAD persistence-path timing breakdown (#2203 U1) loadGraphToLbug is un-timed today; the analyze 'emit' number is the scope-resolution emit bucket, not the CSV->COPY persistence path. Add a zero-cost-when-off per-stage breakdown (csv-emit/copy-nodes/rel-split/ copy-rels/fallback/total + node/rel counts) gated by PROF_LBUG_LOAD=1, mirroring the PROF_SCOPE_RESOLUTION pattern. Document the flag in README. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(lbug): route relationships to per-pair CSVs in the emit pass (#2203 U2) Relationships were written once to a monolithic relations.csv, then re-read line-by-line (regex per edge) and re-split into per-FROM->TO-label-pair files before COPY — writing and reading the entire ~1M-edge set twice. Route each edge to its pair file directly during the single emit pass via a shared RelPairRouter, eliminating the monolithic write + re-read + per-edge regex. The router applies the SAME getNodeLabel + validTables filter as the legacy splitRelCsvByLabelPair, which is retained as a differential oracle. A new differential test asserts the direct-emit per-pair files are byte-for-byte identical to the oracle's, with identical skip/total accounting. The prof line (U1) drops its rel-split stage (routing now folds into csv-emit). Co-Authored-By: Claude Opus 4.8 (1M context) * perf(lbug): skip per-row microtask tick in BufferedCSVWriter (#2203 U3) addRow awaited an already-resolved promise on every buffered row, scheduling a microtask per node even when nothing flushed (millions at scale). It now returns a promise ONLY when it flushes; the node-emit loop awaits once per iteration after the switch. Flush/drain semantics are unchanged, so backpressure on the rows that actually write is preserved and the emitted CSV bytes are byte-identical (covered by the determinism + differential tests). Co-Authored-By: Claude Opus 4.8 (1M context) * bench(lbug): emit throughput + byte-identity gate for the persistence path (#2203 U4) Build-free bench (bench/emit-persistence/measure.mjs) times streamAllCSVsToDisk on a synthetic graph at two scales and gates: (1) an order-independent sha256 fingerprint over every emitted CSV line — the byte-identity guard for the U2/U3 emit optimisations — and (2) a scaling-ratio budget catching an O(n^2) emit re-regression. Wired into ci-tests.yml alongside the cfg/scope-capture benches. The LadybugDB COPY half needs a real DB, so its timing stays in PROF_LBUG_LOAD + the integration round-trip tests (documented in the bench README, with the deferred COPY-parallelism follow-up). Co-Authored-By: Claude Opus 4.8 (1M context) * fix(review): apply autofix feedback (#2203) - P1: router backpressure drain-await rejected with a generic AbortError, masking the real EMFILE/disk-full error. Expose RelPairRouter.lastError and rethrow it in the emit catch — mirrors the oracle's throw streamError ?? err. - P1: cover RelPairRouter error + backpressure + teardown paths with a new unit test (test/unit/rel-pair-routing.test.ts) using an injected mock stream. - P2: wrap streamAllCSVsToDisk body in try/finally so the setMaxListeners bump is always restored (the U2 rel-routing throw path could leak it). - P2: dedup WriteStreamFactory — re-export the canonical type from rel-pair-routing instead of a second identical declaration. - P2: annotate splitRelCsvByLabelPair @internal as the retained differential oracle so a future dead-code sweep doesn't delete the byte-identity guard. - P3: differential test now covers the proc_ prefix + clears GITNEXUS_SORT_GRAPH_OUTPUT to prevent env-leak desync. Co-Authored-By: Claude Opus 4.8 (1M context) * docs(lbug): scope byte-identity to quote-free ids + lock the quote-in-id divergence (#2215 review) The 'byte-identical' claim was unconditional, but the router derives labels from the raw id while the retained splitRelCsvByLabelPair oracle re-derives them via a regex over the escaped row — so for an id containing a double-quote they diverge (the router is the more-correct path). Soften the wording in rel-pair-routing.ts, the bench README, and the differential-test comment to document the exception, and add a differential test asserting the intended divergence (router routes the quote-in-id edge; oracle drops it) so a future change can't silently revert to the buggy regex semantics. Co-Authored-By: Claude Opus 4.8 (1M context) * bench(lbug): per-file fingerprint so the gate catches pair-file mis-routing (#2215 review) fingerprintEmit flattened every line of every per-pair file into one array, sorted globally, and hashed — losing file boundaries, so a row routed to the WRONG pair file produced an identical fingerprint. Hash a per-file digest (filename + sha256(file bytes)) and combine the sorted entry list, so mis-routing (and within-file row reordering) now changes the fingerprint. Baseline regenerated; the new scheme yields a different hash on byte-identical emit, confirming it is sensitive to file structure the old flatten ignored. Co-Authored-By: Claude Opus 4.8 (1M context) * bench(lbug): add absolute large-scale wall-time backstop to the emit gate (#2215 review) The scaling-ratio gate only compares large/small, so a uniform Nx slowdown at both scales passes with ratio ~1.0. Add an opt-in max_ms_large ceiling (1000ms vs observed ~200ms — generous, host-noise-tolerant) that --check enforces alongside the ratio, catching a gross absolute regression the ratio misses. Co-Authored-By: Claude Opus 4.8 (1M context) * test(lbug): cover the sorted-output path in the byte-identity differential (#2215 review) The differential test only exercised the default insertion-order emit path. Add a case under GITNEXUS_SORT_GRAPH_OUTPUT=1 that feeds the oracle the same id-sorted order orderedRelationships() uses and asserts per-pair byte-identity, so within-pair row reordering on the sorted path can't slip past the gate. Co-Authored-By: Claude Opus 4.8 (1M context) * test(lbug): cover the invalid-TO-label skip branch (#2215 review) Only an invalid-FROM label was exercised; the validTables skip is an OR over both endpoints, so the invalid-TO branch was untested (an inverted && would have slipped through). Add a valid-FROM/invalid-TO edge to the differential test and the router unit test, asserting it's skipped identically by router and oracle. Co-Authored-By: Claude Opus 4.8 (1M context) * test(lbug): exercise the BufferedCSVWriter FLUSH_EVERY boundary in vitest (#2215 review) The U3 addRow change (returns a flush promise only on flush; undefined when buffered) and the loop's `if (pending) await pending` were only crossed by the bench, never vitest (all fixtures are <500 nodes). Add a 600-node graph through streamAllCSVsToDisk asserting all rows land exactly once across the 500-row flush boundary — no drops, dups, or corruption. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(lbug): drop redundant step cast in buildRelRow (#2215 review) GraphRelationship.step is already typed number?, so (rel as { step?: number }).step was a no-op structural cast that obscured the shared-type coupling. Use rel.step directly. Byte-identical — bench fingerprint unchanged, differential test green. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(lbug): make the unknown-label node drop explicit (#2215 review) With the U3 `let pending` switch idiom, a node whose label matches neither codeWriterMap nor multiLangWriters left `pending` undefined and was silently dropped — a footgun for a future node type. Add an explicit else with a comment documenting that unknown labels are intentionally not persisted and that a new type must be wired into a writer map. No behavior change (byte-identity + tests unchanged). Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(lbug): drop the unused WriteStreamFactory re-export (#2215 review) The type was re-exported from lbug-adapter 'to preserve this module's surface,' but no external code imports it by name from here (the only test reference is a comment). Keep the import from rel-pair-routing.ts (its canonical home, still used by splitRelCsvByLabelPair's signature) and drop the dead re-export. Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: Claude Opus 4.8 (1M context) --- .github/workflows/ci-tests.yml | 9 + README.md | 1 + gitnexus/bench/emit-persistence/README.md | 58 ++ .../bench/emit-persistence/baselines.json | 6 + gitnexus/bench/emit-persistence/measure.mjs | 216 ++++++ gitnexus/src/core/lbug/csv-generator.ts | 711 ++++++++++-------- gitnexus/src/core/lbug/lbug-adapter.ts | 80 +- gitnexus/src/core/lbug/rel-pair-routing.ts | 159 ++++ .../test/integration/csv-pipeline.test.ts | 294 +++++++- .../test/integration/lbug-load-prof.test.ts | 141 ++++ gitnexus/test/unit/rel-pair-routing.test.ts | 176 +++++ 11 files changed, 1479 insertions(+), 372 deletions(-) create mode 100644 gitnexus/bench/emit-persistence/README.md create mode 100644 gitnexus/bench/emit-persistence/baselines.json create mode 100644 gitnexus/bench/emit-persistence/measure.mjs create mode 100644 gitnexus/src/core/lbug/rel-pair-routing.ts create mode 100644 gitnexus/test/integration/lbug-load-prof.test.ts create mode 100644 gitnexus/test/unit/rel-pair-routing.test.ts diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index ee0d92332..a851696fb 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -276,6 +276,15 @@ jobs: run: node --expose-gc --import tsx bench/cfg/measure.mjs --check working-directory: gitnexus + - name: Emit-persistence throughput / byte-identity guards (#2203) + # Build-free: asserts streamAllCSVsToDisk output is byte-identical + # (order-independent CSV-line fingerprint — the #2203 U2/U3 emit + # optimisations must not change graph content) and that emit wall-time + # stays linear in node+edge count. The LadybugDB COPY half needs a real + # DB, so its timing lives in the runtime PROF_LBUG_LOAD breakdown. + run: node --import tsx bench/emit-persistence/measure.mjs --check + working-directory: gitnexus + - name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial) env: GITNEXUS_BENCH: '1' diff --git a/README.md b/README.md index ab01db810..55e5a03a7 100644 --- a/README.md +++ b/README.md @@ -324,6 +324,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max | `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. | | `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. | | `GITNEXUS_PROFILE_DEFERRED_SLOW_MS` | `3000` (verbose) / `5000` | Per-file threshold in ms above which `processCallsFromExtracted` emits a `slow file …` log line. Parsed via `Number()`: accepts integers (`5000`), scientific notation (`2.5e3`), decimals (`.5`), and hex (`0x10`). Non-finite or non-positive values fall back to the default. | Hunting a few outlier files dominating the deferred call-resolution stage; lower to surface more, raise to focus only on the worst. | +| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. | | `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size `. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. | | `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout ` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. | | `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold `. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. | diff --git a/gitnexus/bench/emit-persistence/README.md b/gitnexus/bench/emit-persistence/README.md new file mode 100644 index 000000000..0a690f40b --- /dev/null +++ b/gitnexus/bench/emit-persistence/README.md @@ -0,0 +1,58 @@ +# Emit-persistence bench (#2203) + +Build-free throughput + byte-identity guard for the **CSV-generation half** of +the graph-DB persistence pipeline (`streamAllCSVsToDisk`), which dominates +large-repo `analyze` wall time alongside parsing (issue #2203). + +```bash +# from gitnexus/ +node --import tsx bench/emit-persistence/measure.mjs # print one JSON line +node --import tsx bench/emit-persistence/measure.mjs --check # gate vs baselines.json +``` + +## What it measures + +A synthetic `KnowledgeGraph` (files + functions + classes + 4 edge types across +the `File→Function`, `File→Class`, `Function→Function` label pairs) at two +scales: + +- **`elapsed_ms_small` / `elapsed_ms_large`** — median wall-clock over `REPS` + runs of `streamAllCSVsToDisk`. +- **`scaling_ratio`** — `(t_large/t_small)/(LARGE/SMALL)`; ~1.0 is linear. The + `--check` gate fails if it exceeds `scaling_budget` (catches an O(n²) + re-regression in the emit/routing path). +- **`fingerprint`** — order-independent sha256 over every emitted CSV line (node + CSVs + per-FROM→TO-label-pair rel CSVs). This is the **byte-identity gate**: + the U2 (direct per-pair routing) and U3 (per-row microtask elimination) + optimisations must not change graph content, and any future change that does + fails `--check`. Byte-identity holds for all quote-free ids; for an id + containing a `"` the router intentionally diverges from — and is more correct + than — the legacy regex oracle (see `src/core/lbug/rel-pair-routing.ts`). + +## What it does NOT measure + +- **The LadybugDB `COPY` half.** Bulk loading needs a live writable DB + connection, so it can't run build-free. Its per-stage timing lives in the + runtime `PROF_LBUG_LOAD=1` breakdown (`[lbug-load prof] csv-emit=… copy-nodes=… + copy-rels=… fallback=… total=…`) and is exercised end-to-end by the + integration round-trip tests (`test/integration/basicblock-roundtrip.test.ts`, + `lbug-core-adapter.test.ts`). +- **Content extraction.** Bench nodes have no backing source files, so the + `content` column is empty — emit cost here reflects the CSV machinery + (routing, escaping, buffering, disk writes), not file reads. +- **At-scale absolute numbers.** The real postgres / kernel-`fs/` wall (issue + #2203's table) is a maintainer-run measurement; this synthetic bench is the + reproducible regression guard, not a substitute for those runs. + +## Deferred follow-up + +Parallelising the `COPY` loop (`PARALLEL=false` is load-bearing; LadybugDB is +single-writer) is **out of scope** for #2203 pending empirical validation of +concurrent-COPY support — the `PROF_LBUG_LOAD` breakdown is the prerequisite +that shows whether COPY is the dominant cost worth that risk. + +## Regenerating the baseline + +```bash +node --import tsx bench/emit-persistence/measure.mjs # copy fingerprint + ratio into baselines.json +``` diff --git a/gitnexus/bench/emit-persistence/baselines.json b/gitnexus/bench/emit-persistence/baselines.json new file mode 100644 index 000000000..93806b252 --- /dev/null +++ b/gitnexus/bench/emit-persistence/baselines.json @@ -0,0 +1,6 @@ +{ + "fingerprint": "1b9dd0b783899b47067c36511d241860f291ac736e57682b0ece14148e3958ff", + "scaling_budget": 1.8, + "max_ms_large": 1000, + "_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted — binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." +} diff --git a/gitnexus/bench/emit-persistence/measure.mjs b/gitnexus/bench/emit-persistence/measure.mjs new file mode 100644 index 000000000..f7623bc34 --- /dev/null +++ b/gitnexus/bench/emit-persistence/measure.mjs @@ -0,0 +1,216 @@ +/** + * Build-free emit-path throughput + byte-identity bench for the graph-DB + * persistence pipeline (issue #2203). + * + * Measures `streamAllCSVsToDisk` — the CSV-generation half of the persistence + * path that U2 (direct per-pair relationship routing) and U3 (per-row + * microtask elimination) optimised. The LadybugDB `COPY` half needs a real DB + * connection, so its timing lives in the runtime `PROF_LBUG_LOAD` breakdown + + * the integration round-trip tests, NOT here (see README.md). + * + * For a synthetic KnowledgeGraph at two scales it reports: + * - elapsed_ms_small / elapsed_ms_large (median over REPS) + a scaling ratio + * `(t_large/t_small)/(LARGE/SMALL)`: ~1.0 linear, ~3.x quadratic; + * - an order-independent sha256 fingerprint over every emitted CSV line + * (node CSVs + per-FROM→TO-label-pair rel CSVs), as the byte-identity gate + * guarding the issue's "byte-identical graph content" requirement. + * + * Build-free: imports the `.ts` hotpaths through tsx + * (`node --import tsx bench/emit-persistence/measure.mjs`). Static `.ts` + * imports work; a top-level `await import()` breaks tsx's lexer. + * + * Without args: prints one JSON object per scenario. + * With `--check`: asserts the fingerprint == the committed baseline AND the + * scaling ratio < the recorded budget; exits non-zero on drift/regression. + */ +import fs from 'node:fs'; +import fsp from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { fileURLToPath } from 'node:url'; + +import { createKnowledgeGraph } from '../../src/core/graph/graph.ts'; +import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.ts'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +// ---- synthetic graph generation (deterministic — no randomness) ---- + +/** + * Build a graph of `entityCount` files, each with 2 functions + 1 class and 4 + * relationships. ids carry valid table-label prefixes (File:/Function:/Class:) + * so edges route to real pairs (File→Function, File→Class, Function→Function) + * — exercising the U2 router across multiple label pairs. Node `content` is + * never populated (no backing files), so emit cost reflects the CSV machinery + * (routing, escaping, buffering, disk writes), not content extraction. + */ +function generateGraph(entityCount) { + const graph = createKnowledgeGraph(); + for (let i = 0; i < entityCount; i++) { + const fp = `src/e${i}.ts`; + const fileId = `File:${fp}`; + const fnA = `Function:${fp}:fnA:1`; + const fnB = `Function:${fp}:fnB:10`; + const cls = `Class:${fp}:C:20`; + graph.addNode({ id: fileId, label: 'File', properties: { name: `e${i}.ts`, filePath: fp } }); + graph.addNode({ + id: fnA, + label: 'Function', + properties: { name: 'fnA', filePath: fp, startLine: 1, endLine: 5, isExported: true }, + }); + graph.addNode({ + id: fnB, + label: 'Function', + properties: { name: 'fnB', filePath: fp, startLine: 10, endLine: 15, isExported: false }, + }); + graph.addNode({ + id: cls, + label: 'Class', + properties: { name: 'C', filePath: fp, startLine: 20, endLine: 30, isExported: true }, + }); + graph.addRelationship({ + id: `${fileId}->${fnA}`, + sourceId: fileId, + targetId: fnA, + type: 'CONTAINS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: `${fileId}->${fnB}`, + sourceId: fileId, + targetId: fnB, + type: 'CONTAINS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: `${fileId}->${cls}`, + sourceId: fileId, + targetId: cls, + type: 'CONTAINS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: `${fnA}->${fnB}`, + sourceId: fnA, + targetId: fnB, + type: 'CALLS', + confidence: 1, + reason: '', + }); + } + return graph; +} + +// ---- byte-identity fingerprint (order-independent) ---- + +/** sha256 over every non-empty line of every emitted CSV file, sorted so the + * digest is a pure function of the emitted line SET (insertion-order agnostic). */ +async function fingerprintEmit(graph, dir) { + await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); + // Per-file digest bound to the filename: a row routed to the WRONG pair file + // (or a header written to the wrong file) changes the fingerprint — a global + // line-flatten could not catch that. File bytes are hashed as-written (so it + // also catches within-file row reordering); the entry list is sorted so + // readdir order doesn't matter. + const entries = []; + for (const name of fs.readdirSync(dir)) { + if (!name.endsWith('.csv')) continue; + const bytes = await fsp.readFile(path.join(dir, name)); + entries.push(`${name}\n${crypto.createHash('sha256').update(bytes).digest('hex')}`); + } + return crypto.createHash('sha256').update(entries.sort().join('\n')).digest('hex'); +} + +// ---- timing ---- + +function median(xs) { + const s = [...xs].sort((a, b) => a - b); + const m = Math.floor(s.length / 2); + return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2; +} + +async function timeEmit(graph, dir, reps) { + await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); // warmup (not counted) + const samples = []; + for (let i = 0; i < reps; i++) { + const start = process.hrtime.bigint(); + await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); + samples.push(Number(process.hrtime.bigint() - start) / 1e6); + } + return median(samples); +} + +const SMALL = 600; +const LARGE = 2400; +const REPS = 5; + +async function measure() { + const tmpRoot = path.join(os.tmpdir(), `gitnexus-emit-bench-${process.pid}`); + await fsp.mkdir(tmpRoot, { recursive: true }); + try { + const smallGraph = generateGraph(SMALL); + const largeGraph = generateGraph(LARGE); + + const fingerprint = await fingerprintEmit(largeGraph, path.join(tmpRoot, 'fp')); + const small = await timeEmit(smallGraph, path.join(tmpRoot, 'small'), REPS); + const large = await timeEmit(largeGraph, path.join(tmpRoot, 'large'), REPS); + const scalingRatio = small > 0 ? large / small / (LARGE / SMALL) : 0; + + return { + scenario: 'streamAllCSVsToDisk', + entities_small: SMALL, + entities_large: LARGE, + nodes_large: LARGE * 4, + rels_large: LARGE * 4, + elapsed_ms_small: Number(small.toFixed(2)), + elapsed_ms_large: Number(large.toFixed(2)), + scaling_ratio: Number(scalingRatio.toFixed(3)), + fingerprint, + }; + } finally { + await fsp.rm(tmpRoot, { recursive: true, force: true }).catch(() => {}); + } +} + +// ---- run ---- + +const CHECK = process.argv.includes('--check'); +const result = await measure(); + +if (!CHECK) { + process.stdout.write(JSON.stringify(result) + '\n'); +} else { + const base = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8')); + const failures = []; + if (result.fingerprint !== base.fingerprint) { + failures.push( + `byte-identity fingerprint drift (got ${result.fingerprint}, expected ${base.fingerprint})`, + ); + } + if (result.scaling_ratio >= base.scaling_budget) { + failures.push( + `scaling ratio ${result.scaling_ratio} >= budget ${base.scaling_budget} ` + + `(${SMALL}->${LARGE} entities, ms ${result.elapsed_ms_small}->${result.elapsed_ms_large})`, + ); + } + // Absolute backstop: the scaling ratio alone passes a uniform Nx slowdown (it + // only compares large/small). A generous, host-noise-tolerant ceiling catches + // a gross absolute regression. Opt-in (only enforced when max_ms_large is set). + if (base.max_ms_large !== undefined && result.elapsed_ms_large >= base.max_ms_large) { + failures.push( + `absolute wall-time regression: elapsed_ms_large ${result.elapsed_ms_large}ms >= budget ` + + `${base.max_ms_large}ms (coarse backstop, not a tight SLA)`, + ); + } + process.stdout.write(JSON.stringify(result) + '\n'); + if (failures.length > 0) { + for (const f of failures) process.stderr.write(`[emit-persistence --check] FAIL: ${f}\n`); + process.exit(1); + } + process.stderr.write('[emit-persistence --check] PASS\n'); +} diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index c7d8413b5..955a2d47c 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -17,7 +17,8 @@ import { createWriteStream, WriteStream } from 'fs'; import path from 'path'; import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; import { KnowledgeGraph } from '../graph/types.js'; -import { NodeTableName } from './schema.js'; +import { NodeTableName, NODE_TABLES } from './schema.js'; +import { RelPairRouter } from './rel-pair-routing.js'; import { parseTruthyEnv } from '../ingestion/utils/env.js'; /** @@ -182,13 +183,20 @@ class BufferedCSVWriter { this.buffer.push(header); } - addRow(row: string) { + /** + * Buffer a row. Returns a promise ONLY when the buffer crossed FLUSH_EVERY + * and a disk write was issued; otherwise returns `undefined` so the caller + * can skip awaiting (#2203 U3) — avoiding a microtask tick on every buffered + * row (millions at scale). The flush promise still resolves on drain, so + * backpressure is preserved on the rows that actually write. + */ + addRow(row: string): Promise | undefined { this.buffer.push(row); this.rows++; if (this.buffer.length >= FLUSH_EVERY) { return this.flush(); } - return Promise.resolve(); + return undefined; } flush(): Promise { @@ -223,10 +231,33 @@ class BufferedCSVWriter { // STREAMING CSV GENERATION — SINGLE PASS // ============================================================================ +/** Canonical relationship CSV header — shared by the emit pass and the + * `splitRelCsvByLabelPair` differential oracle. */ +export const REL_CSV_HEADER = 'from,to,type,confidence,reason,step'; + +/** Build the escaped CSV row (no trailing newline) for one relationship. + * Single source of the relationship row bytes — used by the emit pass and by + * the byte-identity differential test that feeds the legacy split oracle. */ +export const buildRelRow = (rel: GraphRelationship): string => + [ + escapeCSVField(rel.sourceId), + escapeCSVField(rel.targetId), + escapeCSVField(rel.type), + escapeCSVNumber(rel.confidence, 1.0), + escapeCSVField(rel.reason), + escapeCSVNumber(rel.step, 0), + ].join(','); + export interface StreamedCSVResult { nodeFiles: Map; - relCsvPath: string; - relRows: number; + /** pairKey (`From|To`) → per-FROM→TO-label-pair CSV file. */ + relsByPair: Map; + /** Header line shared by every per-pair file. */ + relHeader: string; + /** Edges skipped because an endpoint label is not a valid node table. */ + skippedRels: number; + /** Edges routed to a per-pair file. */ + totalValidRels: number; } /** @@ -253,255 +284,185 @@ export const streamAllCSVsToDisk = async ( const prevMax = process.getMaxListeners(); process.setMaxListeners(prevMax + 40); - const contentCache = new FileContentCache(repoPath); + // try/finally so the listener bump is ALWAYS restored — including the + // rel-routing throw path (#2203 U2) and any node-writer finish() rejection, + // not just the success path (avoids leaking +40 listeners across failed runs + // in long-lived hosts / the test suite). + try { + const contentCache = new FileContentCache(repoPath); - // Create writers for every node type up-front - const fileWriter = new BufferedCSVWriter( - path.join(csvDir, 'file.csv'), - 'id,name,filePath,content', - ); - const folderWriter = new BufferedCSVWriter(path.join(csvDir, 'folder.csv'), 'id,name,filePath'); - const codeElementHeader = 'id,name,filePath,startLine,endLine,isExported,content,description'; - const functionWriter = new BufferedCSVWriter( - path.join(csvDir, 'function.csv'), - codeElementHeader, - ); - const classWriter = new BufferedCSVWriter(path.join(csvDir, 'class.csv'), codeElementHeader); - const interfaceWriter = new BufferedCSVWriter( - path.join(csvDir, 'interface.csv'), - codeElementHeader, - ); - const methodHeader = - 'id,name,filePath,startLine,endLine,isExported,content,description,parameterCount,returnType'; - const methodWriter = new BufferedCSVWriter(path.join(csvDir, 'method.csv'), methodHeader); - const codeElemWriter = new BufferedCSVWriter( - path.join(csvDir, 'codeelement.csv'), - codeElementHeader, - ); - const communityWriter = new BufferedCSVWriter( - path.join(csvDir, 'community.csv'), - 'id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount', - ); - const processWriter = new BufferedCSVWriter( - path.join(csvDir, 'process.csv'), - 'id,label,heuristicLabel,processType,stepCount,communities,entryPointId,terminalId', - ); - - // Section nodes have an extra 'level' column - const sectionWriter = new BufferedCSVWriter( - path.join(csvDir, 'section.csv'), - 'id,name,filePath,startLine,endLine,level,content,description', - ); - - // Route nodes for API endpoint mapping - const routeWriter = new BufferedCSVWriter( - path.join(csvDir, 'route.csv'), - 'id,name,filePath,responseKeys,errorKeys,middleware', - ); - - // Tool nodes for MCP tool definitions - const toolWriter = new BufferedCSVWriter( - path.join(csvDir, 'tool.csv'), - 'id,name,filePath,description', - ); - - // BasicBlock nodes — taint/PDG substrate (issue #2080). No `name` column; - // blocks are identified by id + source span. Emitted by no phase yet. - const basicBlockWriter = new BufferedCSVWriter( - path.join(csvDir, 'basicblock.csv'), - 'id,filePath,startLine,endLine,text', - ); - - // Multi-language node types share the same CSV shape (no isExported column) - const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description'; - const MULTI_LANG_TYPES = [ - 'Struct', - 'Enum', - 'Macro', - 'Typedef', - 'Union', - 'Namespace', - 'Trait', - 'Impl', - 'TypeAlias', - 'Const', - 'Static', - 'Variable', - 'Property', - 'Record', - 'Delegate', - 'Annotation', - 'Constructor', - 'Template', - 'Module', - ] as const; - const propertyHeader = 'id,name,filePath,startLine,endLine,content,description,declaredType'; - const multiLangWriters = new Map(); - for (const t of MULTI_LANG_TYPES) { - multiLangWriters.set( - t, - new BufferedCSVWriter( - path.join(csvDir, `${t.toLowerCase()}.csv`), - t === 'Property' ? propertyHeader : multiLangHeader, - ), + // Create writers for every node type up-front + const fileWriter = new BufferedCSVWriter( + path.join(csvDir, 'file.csv'), + 'id,name,filePath,content', + ); + const folderWriter = new BufferedCSVWriter(path.join(csvDir, 'folder.csv'), 'id,name,filePath'); + const codeElementHeader = 'id,name,filePath,startLine,endLine,isExported,content,description'; + const functionWriter = new BufferedCSVWriter( + path.join(csvDir, 'function.csv'), + codeElementHeader, + ); + const classWriter = new BufferedCSVWriter(path.join(csvDir, 'class.csv'), codeElementHeader); + const interfaceWriter = new BufferedCSVWriter( + path.join(csvDir, 'interface.csv'), + codeElementHeader, + ); + const methodHeader = + 'id,name,filePath,startLine,endLine,isExported,content,description,parameterCount,returnType'; + const methodWriter = new BufferedCSVWriter(path.join(csvDir, 'method.csv'), methodHeader); + const codeElemWriter = new BufferedCSVWriter( + path.join(csvDir, 'codeelement.csv'), + codeElementHeader, + ); + const communityWriter = new BufferedCSVWriter( + path.join(csvDir, 'community.csv'), + 'id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount', + ); + const processWriter = new BufferedCSVWriter( + path.join(csvDir, 'process.csv'), + 'id,label,heuristicLabel,processType,stepCount,communities,entryPointId,terminalId', ); - } - const codeWriterMap: Record = { - Function: functionWriter, - Class: classWriter, - Interface: interfaceWriter, - CodeElement: codeElemWriter, - }; + // Section nodes have an extra 'level' column + const sectionWriter = new BufferedCSVWriter( + path.join(csvDir, 'section.csv'), + 'id,name,filePath,startLine,endLine,level,content,description', + ); - // Deduplicate all node types — the pipeline can produce duplicate IDs across - // all symbol types (Class, Method, Function, etc.), not just File nodes. - // A single Set covering every label prevents PK violations on COPY. - const seenNodeIds = new Set(); + // Route nodes for API endpoint mapping + const routeWriter = new BufferedCSVWriter( + path.join(csvDir, 'route.csv'), + 'id,name,filePath,responseKeys,errorKeys,middleware', + ); - // --- SINGLE PASS over all nodes --- - for (const node of orderedNodes(graph, sortOutput)) { - if (seenNodeIds.has(node.id)) continue; - seenNodeIds.add(node.id); + // Tool nodes for MCP tool definitions + const toolWriter = new BufferedCSVWriter( + path.join(csvDir, 'tool.csv'), + 'id,name,filePath,description', + ); - switch (node.label) { - case 'File': { - const content = await extractContent(node, contentCache); - await fileWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVField(content), - ].join(','), - ); - break; - } - case 'Folder': - await folderWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - ].join(','), - ); - break; - case 'Community': { - const keywords = node.properties.keywords || []; - const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/\\/g, '\\\\').replace(/'/g, "''").replace(/,/g, '\\,')}'`).join(',')}]`; - await communityWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.heuristicLabel || ''), - keywordsStr, - escapeCSVField(node.properties.description || ''), - escapeCSVField(node.properties.enrichedBy || 'heuristic'), - escapeCSVNumber(node.properties.cohesion, 0), - escapeCSVNumber(node.properties.symbolCount, 0), - ].join(','), - ); - break; - } - case 'Process': { - const communities = node.properties.communities || []; - const communitiesStr = `[${communities.map((c: string) => `'${c.replace(/'/g, "''")}'`).join(',')}]`; - await processWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.heuristicLabel || ''), - escapeCSVField(node.properties.processType || ''), - escapeCSVNumber(node.properties.stepCount, 0), - escapeCSVField(communitiesStr), - escapeCSVField(node.properties.entryPointId || ''), - escapeCSVField(node.properties.terminalId || ''), - ].join(','), - ); - break; - } - case 'Method': { - const content = await extractContent(node, contentCache); - await methodWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVNumber(node.properties.startLine, -1), - escapeCSVNumber(node.properties.endLine, -1), - node.properties.isExported ? 'true' : 'false', - escapeCSVField(content), - escapeCSVField(node.properties.description || ''), - escapeCSVNumber(node.properties.parameterCount, 0), - escapeCSVField(node.properties.returnType || ''), - ].join(','), - ); - break; - } - case 'Section': { - const content = await extractContent(node, contentCache); - await sectionWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVNumber(node.properties.startLine, -1), - escapeCSVNumber(node.properties.endLine, -1), - escapeCSVNumber(node.properties.level, 1), - escapeCSVField(content), - escapeCSVField(node.properties.description || ''), - ].join(','), - ); - break; - } - case 'Route': { - const responseKeys = node.properties.responseKeys || []; - // LadybugDB array literal inside a quoted CSV field: escapeCSVField wraps in "..." - // and the array uses single-quoted elements - const keysStr = `[${responseKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`; - const errorKeys = node.properties.errorKeys || []; - const errorKeysStr = `[${errorKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`; - const middleware = node.properties.middleware || []; - const middlewareStr = `[${middleware.map((m: string) => `'${m.replace(/'/g, "''")}'`).join(',')}]`; - await routeWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVField(keysStr), - escapeCSVField(errorKeysStr), - escapeCSVField(middlewareStr), - ].join(','), - ); - break; - } - case 'Tool': - await toolWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.name || ''), - escapeCSVField(node.properties.filePath || ''), - escapeCSVField(node.properties.description || ''), - ].join(','), - ); - break; - case 'BasicBlock': - await basicBlockWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.filePath || ''), - escapeCSVNumber(node.properties.startLine, -1), - escapeCSVNumber(node.properties.endLine, -1), - escapeCSVField(node.properties.text || ''), - ].join(','), - ); - break; - default: { - // Code element nodes (Function, Class, Interface, CodeElement) - const writer = codeWriterMap[node.label]; - if (writer) { + // BasicBlock nodes — taint/PDG substrate (issue #2080). No `name` column; + // blocks are identified by id + source span. Emitted by no phase yet. + const basicBlockWriter = new BufferedCSVWriter( + path.join(csvDir, 'basicblock.csv'), + 'id,filePath,startLine,endLine,text', + ); + + // Multi-language node types share the same CSV shape (no isExported column) + const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description'; + const MULTI_LANG_TYPES = [ + 'Struct', + 'Enum', + 'Macro', + 'Typedef', + 'Union', + 'Namespace', + 'Trait', + 'Impl', + 'TypeAlias', + 'Const', + 'Static', + 'Variable', + 'Property', + 'Record', + 'Delegate', + 'Annotation', + 'Constructor', + 'Template', + 'Module', + ] as const; + const propertyHeader = 'id,name,filePath,startLine,endLine,content,description,declaredType'; + const multiLangWriters = new Map(); + for (const t of MULTI_LANG_TYPES) { + multiLangWriters.set( + t, + new BufferedCSVWriter( + path.join(csvDir, `${t.toLowerCase()}.csv`), + t === 'Property' ? propertyHeader : multiLangHeader, + ), + ); + } + + const codeWriterMap: Record = { + Function: functionWriter, + Class: classWriter, + Interface: interfaceWriter, + CodeElement: codeElemWriter, + }; + + // Deduplicate all node types — the pipeline can produce duplicate IDs across + // all symbol types (Class, Method, Function, etc.), not just File nodes. + // A single Set covering every label prevents PK violations on COPY. + const seenNodeIds = new Set(); + + // --- SINGLE PASS over all nodes --- + for (const node of orderedNodes(graph, sortOutput)) { + if (seenNodeIds.has(node.id)) continue; + seenNodeIds.add(node.id); + + // addRow returns a promise only when it flushes; awaiting it once after the + // switch (instead of `await`-ing every addRow) skips a per-row microtask + // tick on the ~FLUSH_EVERY-1 buffered rows between flushes (#2203 U3). + let pending: Promise | undefined; + switch (node.label) { + case 'File': { const content = await extractContent(node, contentCache); - await writer.addRow( + pending = fileWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVField(content), + ].join(','), + ); + break; + } + case 'Folder': + pending = folderWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + ].join(','), + ); + break; + case 'Community': { + const keywords = node.properties.keywords || []; + const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/\\/g, '\\\\').replace(/'/g, "''").replace(/,/g, '\\,')}'`).join(',')}]`; + pending = communityWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.heuristicLabel || ''), + keywordsStr, + escapeCSVField(node.properties.description || ''), + escapeCSVField(node.properties.enrichedBy || 'heuristic'), + escapeCSVNumber(node.properties.cohesion, 0), + escapeCSVNumber(node.properties.symbolCount, 0), + ].join(','), + ); + break; + } + case 'Process': { + const communities = node.properties.communities || []; + const communitiesStr = `[${communities.map((c: string) => `'${c.replace(/'/g, "''")}'`).join(',')}]`; + pending = processWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.heuristicLabel || ''), + escapeCSVField(node.properties.processType || ''), + escapeCSVNumber(node.properties.stepCount, 0), + escapeCSVField(communitiesStr), + escapeCSVField(node.properties.entryPointId || ''), + escapeCSVField(node.properties.terminalId || ''), + ].join(','), + ); + break; + } + case 'Method': { + const content = await extractContent(node, contentCache); + pending = methodWriter.addRow( [ escapeCSVField(node.id), escapeCSVField(node.properties.name || ''), @@ -511,101 +472,199 @@ export const streamAllCSVsToDisk = async ( node.properties.isExported ? 'true' : 'false', escapeCSVField(content), escapeCSVField(node.properties.description || ''), + escapeCSVNumber(node.properties.parameterCount, 0), + escapeCSVField(node.properties.returnType || ''), ].join(','), ); - } else { - // Multi-language node types (Struct, Impl, Trait, Macro, etc.) - const mlWriter = multiLangWriters.get(node.label); - if (mlWriter) { + break; + } + case 'Section': { + const content = await extractContent(node, contentCache); + pending = sectionWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVNumber(node.properties.startLine, -1), + escapeCSVNumber(node.properties.endLine, -1), + escapeCSVNumber(node.properties.level, 1), + escapeCSVField(content), + escapeCSVField(node.properties.description || ''), + ].join(','), + ); + break; + } + case 'Route': { + const responseKeys = node.properties.responseKeys || []; + // LadybugDB array literal inside a quoted CSV field: escapeCSVField wraps in "..." + // and the array uses single-quoted elements + const keysStr = `[${responseKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`; + const errorKeys = node.properties.errorKeys || []; + const errorKeysStr = `[${errorKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`; + const middleware = node.properties.middleware || []; + const middlewareStr = `[${middleware.map((m: string) => `'${m.replace(/'/g, "''")}'`).join(',')}]`; + pending = routeWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVField(keysStr), + escapeCSVField(errorKeysStr), + escapeCSVField(middlewareStr), + ].join(','), + ); + break; + } + case 'Tool': + pending = toolWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVField(node.properties.description || ''), + ].join(','), + ); + break; + case 'BasicBlock': + pending = basicBlockWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.filePath || ''), + escapeCSVNumber(node.properties.startLine, -1), + escapeCSVNumber(node.properties.endLine, -1), + escapeCSVField(node.properties.text || ''), + ].join(','), + ); + break; + default: { + // Code element nodes (Function, Class, Interface, CodeElement) + const writer = codeWriterMap[node.label]; + if (writer) { const content = await extractContent(node, contentCache); - await mlWriter.addRow( + pending = writer.addRow( [ escapeCSVField(node.id), escapeCSVField(node.properties.name || ''), escapeCSVField(node.properties.filePath || ''), escapeCSVNumber(node.properties.startLine, -1), escapeCSVNumber(node.properties.endLine, -1), + node.properties.isExported ? 'true' : 'false', escapeCSVField(content), escapeCSVField(node.properties.description || ''), - ...(node.label === 'Property' - ? [escapeCSVField(node.properties.declaredType || '')] - : []), ].join(','), ); + } else { + // Multi-language node types (Struct, Impl, Trait, Macro, etc.) + const mlWriter = multiLangWriters.get(node.label); + if (mlWriter) { + const content = await extractContent(node, contentCache); + pending = mlWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.name || ''), + escapeCSVField(node.properties.filePath || ''), + escapeCSVNumber(node.properties.startLine, -1), + escapeCSVNumber(node.properties.endLine, -1), + escapeCSVField(content), + escapeCSVField(node.properties.description || ''), + ...(node.label === 'Property' + ? [escapeCSVField(node.properties.declaredType || '')] + : []), + ].join(','), + ); + } else { + // Unknown label: not in codeWriterMap or multiLangWriters, so there + // is no CSV table for it and it is intentionally NOT persisted — + // `pending` stays undefined, so the loop awaits nothing. Made + // explicit so a future node type isn't silently dropped here: wire + // it into one of the writer maps above (or this branch). + } } + break; } - break; + } + if (pending) await pending; + } + + // Finish all node writers + const allWriters = [ + fileWriter, + folderWriter, + functionWriter, + classWriter, + interfaceWriter, + methodWriter, + codeElemWriter, + communityWriter, + processWriter, + sectionWriter, + routeWriter, + toolWriter, + basicBlockWriter, + ...multiLangWriters.values(), + ]; + await Promise.all(allWriters.map((w) => w.finish())); + + // --- Stream relationships directly to per-FROM→TO-label-pair files --- + // (#2203 U2) Route every edge to its pair file in this single pass. The old + // monolithic relations.csv — and its line-by-line re-read + per-edge regex + // re-split in loadGraphToLbug — are gone, so the ~1M-edge set is written and + // read once instead of twice. The router applies the SAME label-derivation + + // validTables filter as the legacy splitRelCsvByLabelPair, so the per-pair + // files are byte-identical (asserted by the differential test). + const relRouter = new RelPairRouter(csvDir, REL_CSV_HEADER, new Set(NODE_TABLES)); + try { + for (const rel of orderedRelationships(graph, sortOutput)) { + const pending = relRouter.route(rel.sourceId, rel.targetId, buildRelRow(rel)); + if (pending) await pending; + } + await relRouter.close(); + } catch (err) { + relRouter.destroy(); + // Rethrow the real stream error (EMFILE / disk-full) rather than the generic + // AbortError a pending drain-await rejects with — mirrors the retained + // splitRelCsvByLabelPair's `throw streamError ?? err`. + throw relRouter.lastError ?? err; + } + + // Build result map — only include tables that have rows + const nodeFiles = new Map(); + const tableMap: [NodeTableName, BufferedCSVWriter][] = [ + ['File', fileWriter], + ['Folder', folderWriter], + ['Function', functionWriter], + ['Class', classWriter], + ['Interface', interfaceWriter], + ['Method', methodWriter], + ['CodeElement', codeElemWriter], + ['Community', communityWriter], + ['Process', processWriter], + ['Section' as NodeTableName, sectionWriter], + ['Route' as NodeTableName, routeWriter], + ['Tool' as NodeTableName, toolWriter], + ['BasicBlock' as NodeTableName, basicBlockWriter], + ...Array.from(multiLangWriters.entries()).map( + ([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter], + ), + ]; + for (const [name, writer] of tableMap) { + if (writer.rows > 0) { + nodeFiles.set(name, { + csvPath: path.join(csvDir, `${name.toLowerCase()}.csv`), + rows: writer.rows, + }); } } + + return { + nodeFiles, + relsByPair: relRouter.byPair, + relHeader: REL_CSV_HEADER, + skippedRels: relRouter.skipped, + totalValidRels: relRouter.total, + }; + } finally { + // Restore original process listener limit on every path (success or throw). + process.setMaxListeners(prevMax); } - - // Finish all node writers - const allWriters = [ - fileWriter, - folderWriter, - functionWriter, - classWriter, - interfaceWriter, - methodWriter, - codeElemWriter, - communityWriter, - processWriter, - sectionWriter, - routeWriter, - toolWriter, - basicBlockWriter, - ...multiLangWriters.values(), - ]; - await Promise.all(allWriters.map((w) => w.finish())); - - // --- Stream relationship CSV --- - const relCsvPath = path.join(csvDir, 'relations.csv'); - const relWriter = new BufferedCSVWriter(relCsvPath, 'from,to,type,confidence,reason,step'); - for (const rel of orderedRelationships(graph, sortOutput)) { - await relWriter.addRow( - [ - escapeCSVField(rel.sourceId), - escapeCSVField(rel.targetId), - escapeCSVField(rel.type), - escapeCSVNumber(rel.confidence, 1.0), - escapeCSVField(rel.reason), - escapeCSVNumber((rel as any).step, 0), - ].join(','), - ); - } - await relWriter.finish(); - - // Build result map — only include tables that have rows - const nodeFiles = new Map(); - const tableMap: [NodeTableName, BufferedCSVWriter][] = [ - ['File', fileWriter], - ['Folder', folderWriter], - ['Function', functionWriter], - ['Class', classWriter], - ['Interface', interfaceWriter], - ['Method', methodWriter], - ['CodeElement', codeElemWriter], - ['Community', communityWriter], - ['Process', processWriter], - ['Section' as NodeTableName, sectionWriter], - ['Route' as NodeTableName, routeWriter], - ['Tool' as NodeTableName, toolWriter], - ['BasicBlock' as NodeTableName, basicBlockWriter], - ...Array.from(multiLangWriters.entries()).map( - ([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter], - ), - ]; - for (const [name, writer] of tableMap) { - if (writer.rows > 0) { - nodeFiles.set(name, { - csvPath: path.join(csvDir, `${name.toLowerCase()}.csv`), - rows: writer.rows, - }); - } - } - - // Restore original process listener limit - process.setMaxListeners(prevMax); - - return { nodeFiles, relCsvPath, relRows: relWriter.rows }; }; diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index 158d324e8..457c49175 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -19,6 +19,7 @@ import { NodeTableName, } from './schema.js'; import { streamAllCSVsToDisk } from './csv-generator.js'; +import { getNodeLabel as deriveNodeLabel, type WriteStreamFactory } from './rel-pair-routing.js'; import type { CachedEmbedding } from '../embeddings/types.js'; import { extensionManager, type ExtensionEnsureOptions } from './extension-loader.js'; import { @@ -48,9 +49,9 @@ import { logger } from '../logger.js'; // --------------------------------------------------------------------------- // Relationship CSV splitting — extracted for testability (PR #818) // --------------------------------------------------------------------------- - -/** Factory for creating WriteStreams — injectable for testing. */ -export type WriteStreamFactory = (filePath: string) => import('fs').WriteStream; +// WriteStreamFactory is imported above from rel-pair-routing.ts (its canonical +// home) for splitRelCsvByLabelPair's signature; no external code imports it from +// here, so it is not re-exported. /** Result of splitting the relationship CSV into per-label-pair files. */ export interface RelCsvSplitResult { @@ -64,6 +65,15 @@ export interface RelCsvSplitResult { /** * Split a relationship CSV into per-label-pair files on disk. * + * @internal RETAINED AS A DIFFERENTIAL ORACLE. As of #2203 U2, production emit + * routes relationships to per-pair files directly during the single pass (see + * RelPairRouter in `rel-pair-routing.ts`), so this function has NO production + * callers — it is kept ONLY so the byte-identity test in + * `test/integration/csv-pipeline.test.ts` ("direct per-pair emit matches the + * split oracle") can diff the direct-emit output against this proven path. Do + * NOT delete it as dead code without also removing that test and accepting the + * loss of the byte-identity guard (and likewise `test/unit/rel-csv-split.test.ts`). + * * Streams the CSV line-by-line, routing each relationship to a file named * `rel_{fromLabel}_{toLabel}.csv`. Handles backpressure correctly: only one * drain listener per stream at a time, and readline resumes only when ALL @@ -878,6 +888,17 @@ export const loadGraphToLbug = async ( const log = onProgress || (() => {}); + // ── #2203 persistence-path profiling ────────────────────────────────── + // Mirrors the PROF_SCOPE_RESOLUTION pattern (scope-resolution/pipeline/ + // run.ts): zero-cost when off — process.hrtime.bigint() is only read under + // PROF_LBUG_LOAD=1, and the summary is logged behind the same gate. Fills + // the gap that the DB-persistence path is un-timed today (the analyze + // "emit" number is the scope-resolution emit bucket, not this COPY path). + const PROF = process.env.PROF_LBUG_LOAD === '1'; + const mark = (): bigint => (PROF ? process.hrtime.bigint() : 0n); + const span = (a: bigint, b: bigint): string => (Number(b - a) / 1e6).toFixed(1); + const tStart = mark(); + let csvDir: string; if (process.platform === 'win32' && /[^\x00-\x7F]/.test(storagePath)) { const hash = crypto.createHash('sha256').update(storagePath).digest('hex').slice(0, 16); @@ -888,13 +909,9 @@ export const loadGraphToLbug = async ( log('Streaming CSVs to disk...'); const csvResult = await streamAllCSVsToDisk(graph, repoPath, csvDir); + const tCsv = mark(); const validTables = new Set(NODE_TABLES as readonly string[]); - const getNodeLabel = (nodeId: string): string => { - if (nodeId.startsWith('comm_')) return 'Community'; - if (nodeId.startsWith('proc_')) return 'Process'; - return nodeId.split(':')[0]; - }; // Bulk COPY all node CSVs (sequential — LadybugDB allows only one write txn at a time) const nodeFiles = [...csvResult.nodeFiles.entries()]; @@ -924,37 +941,32 @@ export const loadGraphToLbug = async ( } } - // Bulk COPY relationships — split by FROM→TO label pair (LadybugDB requires it) - const { relHeader, relsByPairMeta, pairWriteStreams, skippedRels, totalValidRels } = - await splitRelCsvByLabelPair(csvResult.relCsvPath, csvDir, validTables, getNodeLabel); + const tCopyNodes = mark(); - // Close all per-pair write streams before COPY. `stream/promises.finished` - // resolves on the stream's 'finish' event and rejects on 'error' — replaces - // a hand-rolled promisification with the stdlib primitive. - await Promise.all( - Array.from(pairWriteStreams.values()).map(async (ws) => { - ws.end(); - await finished(ws); - }), - ); + // Bulk COPY relationships. They were already routed to per-FROM→TO-label-pair + // files during the emit pass (#2203 U2) — there is no monolithic relations.csv + // to re-read/re-split here; we COPY each pair file directly. + const { relsByPair, relHeader, skippedRels, totalValidRels } = csvResult; + let tCopyRels = tCopyNodes; + let tFallback = tCopyNodes; const insertedRels = totalValidRels; const warnings: string[] = []; if (insertedRels > 0) { - log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPairMeta.size} types`); + log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPair.size} types`); let pairIdx = 0; let failedPairEdges = 0; const failedPairCsvPaths = new Set(); - for (const [pairKey, { csvPath: pairCsvPath, rows }] of relsByPairMeta) { + for (const [pairKey, { csvPath: pairCsvPath, rows }] of relsByPair) { pairIdx++; const [fromLabel, toLabel] = pairKey.split('|'); const normalizedPath = normalizeCopyPath(pairCsvPath); const copyQuery = `COPY ${REL_TABLE_NAME} FROM "${normalizedPath}" (from="${fromLabel}", to="${toLabel}", HEADER=true, ESCAPE='"', DELIM=',', QUOTE='"', PARALLEL=false, auto_detect=false)`; if (pairIdx % 5 === 0 || rows > 1000) { - log(`Loading edges: ${pairIdx}/${relsByPairMeta.size} types (${fromLabel} -> ${toLabel})`); + log(`Loading edges: ${pairIdx}/${relsByPair.size} types (${fromLabel} -> ${toLabel})`); } try { @@ -980,6 +992,7 @@ export const loadGraphToLbug = async ( } catch {} } } + tCopyRels = mark(); if (failedPairCsvPaths.size > 0) { log(`Inserting ${failedPairEdges} edges individually (missing schema pairs)`); @@ -999,15 +1012,14 @@ export const loadGraphToLbug = async ( } catch {} } if (allLines.length > 1) { - await fallbackRelationshipInserts(allLines, validTables, getNodeLabel); + await fallbackRelationshipInserts(allLines, validTables, deriveNodeLabel); } } + tFallback = mark(); } - // Cleanup all CSVs - try { - await fs.unlink(csvResult.relCsvPath); - } catch {} + // Cleanup all CSVs (per-pair rel files are unlinked in the COPY loop above; + // the remaining sweep below catches node CSVs + any leftover pair files). for (const [, { csvPath }] of csvResult.nodeFiles) { try { await fs.unlink(csvPath); @@ -1025,6 +1037,18 @@ export const loadGraphToLbug = async ( await fs.rmdir(csvDir); } catch {} + if (PROF) { + const tEnd = mark(); + let totalNodeRows = 0; + for (const [, { rows }] of csvResult.nodeFiles) totalNodeRows += rows; + logger.warn( + `[lbug-load prof] csv-emit=${span(tStart, tCsv)}ms ` + + `copy-nodes=${span(tCsv, tCopyNodes)}ms copy-rels=${span(tCopyNodes, tCopyRels)}ms ` + + `fallback=${span(tCopyRels, tFallback)}ms total=${span(tStart, tEnd)}ms ` + + `(${totalNodeRows} nodes, ${insertedRels} rels)`, + ); + } + return { success: true, insertedRels, skippedRels, warnings }; }; diff --git a/gitnexus/src/core/lbug/rel-pair-routing.ts b/gitnexus/src/core/lbug/rel-pair-routing.ts new file mode 100644 index 000000000..a10a817fa --- /dev/null +++ b/gitnexus/src/core/lbug/rel-pair-routing.ts @@ -0,0 +1,159 @@ +/** + * Relationship per-label-pair routing (#2203 U2). + * + * LadybugDB's bulk `COPY` into the single `CodeRelation` rel table requires a + * separate CSV per FROM→TO node-label pair (the `from=`/`to=` COPY params). + * Historically the emit pass wrote one monolithic `relations.csv`, which + * `loadGraphToLbug` then RE-READ line-by-line (regex per edge) and re-split + * into per-pair files — writing and reading the entire ~1M-edge set twice. + * + * This router lets the single emit pass route each edge to its per-pair file + * directly, so the monolithic write + re-read + per-edge regex are all gone. + * The label-derivation + validTables filtering + per-pair-file format here match + * the legacy `splitRelCsvByLabelPair`, so the per-pair files are byte-identical + * for all quote-free ids — see the differential test in + * `test/integration/csv-pipeline.test.ts`. ONE intentional divergence: this + * router derives the label from the RAW id, while the oracle re-derives it via a + * regex over the ESCAPED row — so for an id containing a `"` the router is the + * more-correct path (it routes the edge to the right pair; the oracle's regex + * mis-buckets or drops it). `splitRelCsvByLabelPair` is retained as the + * differential oracle (the quote-in-id divergence is asserted explicitly). + * + * Backpressure: at most one stream is awaited at a time (the caller routes + * edges sequentially and awaits the returned drain promise before the next), + * mirroring the legacy split's `for await` invariant. The hot path (existing + * pair, no backpressure) returns `void` — no microtask per edge. + */ +import path from 'path'; +import { createWriteStream, type WriteStream } from 'fs'; +import { once } from 'events'; +import { finished } from 'stream/promises'; + +/** Injectable for tests (backpressure/error simulation), mirroring split. */ +export type WriteStreamFactory = (filePath: string) => WriteStream; + +/** + * Derive a node's table label from its graph id. Matches the legacy + * `getNodeLabel` that lived inline in `loadGraphToLbug`: + * - `comm_*` → Community + * - `proc_*` → Process + * - otherwise the prefix before the first `:` (e.g. `Function:…` → Function) + */ +export const getNodeLabel = (nodeId: string): string => { + if (nodeId.startsWith('comm_')) return 'Community'; + if (nodeId.startsWith('proc_')) return 'Process'; + return nodeId.split(':')[0]; +}; + +export interface RelPairMeta { + csvPath: string; + rows: number; +} + +/** + * Routes already-escaped relationship CSV rows to per-FROM→TO-label-pair + * files. Filters edges whose endpoint labels are not valid node tables + * (counted as `skipped`), exactly as the legacy split did. + */ +export class RelPairRouter { + /** pairKey (`From|To`) → { csvPath, rows } */ + readonly byPair = new Map(); + private readonly streams = new Map(); + skipped = 0; + total = 0; + + private streamError: Error | null = null; + private readonly abort = new AbortController(); + + constructor( + private readonly csvDir: string, + private readonly header: string, + private readonly validTables: Set, + private readonly wsFactory: WriteStreamFactory = (p) => createWriteStream(p, 'utf-8'), + ) {} + + private markError = (err: Error): void => { + this.streamError ??= err; + this.abort.abort(err); + }; + + /** + * The first stream error observed, if any. Lets the emit caller rethrow the + * real error (EMFILE / disk-full) instead of the generic `AbortError` that a + * pending `once(ws,'drain',{signal})` rejects with when the abort fires — + * mirroring the retained `splitRelCsvByLabelPair`'s `throw streamError ?? err`. + */ + get lastError(): Error | null { + return this.streamError; + } + + /** + * Route one already-escaped CSV row (no trailing newline) to its pair file. + * Returns `void` on the synchronous hot path; a `Promise` only when a + * stream signals backpressure (or a new pair's header does) — the caller + * awaits the promise before routing the next edge. + */ + route(fromId: string, toId: string, row: string): void | Promise { + if (this.streamError) throw this.streamError; + + const fromLabel = getNodeLabel(fromId); + const toLabel = getNodeLabel(toId); + if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) { + this.skipped++; + return; + } + + const pairKey = `${fromLabel}|${toLabel}`; + const ws = this.streams.get(pairKey); + if (ws === undefined) { + // First edge for this pair: open the stream, write header + row. + return this.openAndWrite(pairKey, fromLabel, toLabel, row); + } + + this.byPair.get(pairKey)!.rows++; + this.total++; + if (!ws.write(row + '\n')) { + return once(ws, 'drain', { signal: this.abort.signal }).then(() => undefined); + } + } + + private async openAndWrite( + pairKey: string, + fromLabel: string, + toLabel: string, + row: string, + ): Promise { + const csvPath = path.join(this.csvDir, `rel_${fromLabel}_${toLabel}.csv`); + const ws = this.wsFactory(csvPath); + ws.on('error', this.markError); + this.streams.set(pairKey, ws); + this.byPair.set(pairKey, { csvPath, rows: 1 }); + this.total++; + if (!ws.write(this.header + '\n')) { + await once(ws, 'drain', { signal: this.abort.signal }); + } + if (!ws.write(row + '\n')) { + await once(ws, 'drain', { signal: this.abort.signal }); + } + } + + /** Flush + close every pair stream. Rejects if any stream errored. */ + async close(): Promise { + if (this.streamError) { + this.destroy(); + throw this.streamError; + } + await Promise.all( + Array.from(this.streams.values()).map(async (ws) => { + ws.end(); + await finished(ws); + }), + ); + if (this.streamError) throw this.streamError; + } + + /** Tear down all streams (no flush) — used on the error path. */ + destroy(): void { + for (const ws of this.streams.values()) ws.destroy(); + } +} diff --git a/gitnexus/test/integration/csv-pipeline.test.ts b/gitnexus/test/integration/csv-pipeline.test.ts index cf44c85b2..1d3ee0e7c 100644 --- a/gitnexus/test/integration/csv-pipeline.test.ts +++ b/gitnexus/test/integration/csv-pipeline.test.ts @@ -4,17 +4,45 @@ * Tests: streamAllCSVsToDisk with real graph data. * Covers hardening fixes: LRU cache (#24), BufferedCSVWriter flush */ -import { describe, it, expect, beforeAll, afterAll } from 'vitest'; +import { describe, it, expect, beforeAll, beforeEach, afterAll } from 'vitest'; import fs from 'fs/promises'; +import { finished } from 'stream/promises'; import path from 'path'; import { createTempDir, type TestDBHandle } from '../helpers/test-db.js'; import { buildTestGraph, type TestNodeInput, type TestRelInput } from '../helpers/test-graph.js'; -import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.js'; +import { + streamAllCSVsToDisk, + buildRelRow, + REL_CSV_HEADER, +} from '../../src/core/lbug/csv-generator.js'; +import { splitRelCsvByLabelPair } from '../../src/core/lbug/lbug-adapter.js'; +import { getNodeLabel } from '../../src/core/lbug/rel-pair-routing.js'; +import { NODE_TABLES } from '../../src/core/lbug/schema.js'; let tmpHandle: TestDBHandle; let csvDir: string; let repoDir: string; +/** Data rows (header dropped) of one CSV file's text. */ +const dataRowsOf = (csv: string): string[] => + csv + .trim() + .split('\n') + .slice(1) + .filter((l) => l.length > 0); + +/** Concatenate data rows from every per-pair rel file (#2203 U2), pair keys + * sorted so the concatenation order is deterministic regardless of map order. */ +const readAllRelRows = async ( + relsByPair: Map, +): Promise => { + const rows: string[] = []; + for (const key of [...relsByPair.keys()].sort()) { + rows.push(...dataRowsOf(await fs.readFile(relsByPair.get(key)!.csvPath, 'utf-8'))); + } + return rows; +}; + beforeAll(async () => { tmpHandle = await createTempDir('csv-pipeline-test-'); csvDir = path.join(tmpHandle.dbPath, 'csv'); @@ -76,9 +104,9 @@ describe('streamAllCSVsToDisk', () => { { id: 'folder:src', label: 'Folder', name: 'src', filePath: 'src' }, ], [ - { sourceId: 'func:main', targetId: 'func:helper', type: 'CALLS' }, - { sourceId: 'file:src/index.ts', targetId: 'func:main', type: 'CONTAINS' }, - { sourceId: 'file:src/utils.ts', targetId: 'func:helper', type: 'CONTAINS' }, + { sourceId: 'Function:main', targetId: 'Function:helper', type: 'CALLS' }, + { sourceId: 'File:src/index.ts', targetId: 'Function:main', type: 'CONTAINS' }, + { sourceId: 'File:src/utils.ts', targetId: 'Function:helper', type: 'CONTAINS' }, ], ); @@ -86,7 +114,8 @@ describe('streamAllCSVsToDisk', () => { // Check that CSV files were created expect(result.nodeFiles.size).toBeGreaterThan(0); - expect(result.relRows).toBe(3); + expect(result.totalValidRels).toBe(3); + expect(result.skippedRels).toBe(0); // Verify File CSV const fileCsv = result.nodeFiles.get('File'); @@ -108,10 +137,12 @@ describe('streamAllCSVsToDisk', () => { expect(folderCsv).toBeDefined(); expect(folderCsv!.rows).toBe(1); - // Verify relations CSV exists - const relContent = await fs.readFile(result.relCsvPath, 'utf-8'); - const relLines = relContent.trim().split('\n'); - expect(relLines.length).toBe(4); // header + 3 relationships + // Relationships are routed to per-FROM→TO-label-pair files (#2203 U2): + // Function→Function (CALLS) + File→Function (2× CONTAINS). + expect(result.relsByPair.has('Function|Function')).toBe(true); + expect(result.relsByPair.has('File|Function')).toBe(true); + expect(result.relsByPair.get('File|Function')!.rows).toBe(2); + expect(await readAllRelRows(result.relsByPair)).toHaveLength(3); }); it('CSV content is properly escaped', async () => { @@ -210,7 +241,8 @@ describe('streamAllCSVsToDisk', () => { const graph = buildTestGraph([], []); const result = await streamAllCSVsToDisk(graph, repoDir, csvDir); expect(result.nodeFiles.size).toBe(0); - expect(result.relRows).toBe(0); + expect(result.totalValidRels).toBe(0); + expect(result.relsByPair.size).toBe(0); }); it('handles node with empty string properties', async () => { @@ -221,6 +253,27 @@ describe('streamAllCSVsToDisk', () => { expect(fileCsv).toBeDefined(); expect(fileCsv!.rows).toBe(1); }); + + it('crosses the BufferedCSVWriter FLUSH_EVERY boundary without losing rows', async () => { + // FLUSH_EVERY=500; a >500-node graph forces ≥1 mid-stream flush, exercising + // addRow's flush-promise return + the loop's `if (pending) await pending` + // path that the small fixtures above never reach (only the bench did). + const N = 600; + const nodes = Array.from({ length: N }, (_, i) => ({ + id: `File:src/f${i}.ts`, + label: 'File' as const, + name: `f${i}.ts`, + filePath: `src/f${i}.ts`, + })); + const result = await streamAllCSVsToDisk(buildTestGraph(nodes), repoDir, csvDir); + + const fileCsv = result.nodeFiles.get('File'); + expect(fileCsv).toBeDefined(); + expect(fileCsv!.rows).toBe(N); // no rows dropped/duplicated at the flush boundary + const dataRows = dataRowsOf(await fs.readFile(fileCsv!.csvPath, 'utf-8')); + expect(dataRows).toHaveLength(N); + expect(new Set(dataRows).size).toBe(N); // all distinct — no flush-boundary corruption + }); }); /** @@ -234,15 +287,18 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => { // Folder nodes: single-line CSV rows (no multi-line `content` column), so the // id is the first comma-separated field and split('\n') is safe. ids are // deliberately NOT in insertion order (c, a, b). + // ids use the `Folder:` prefix so getNodeLabel derives the valid `Folder` + // table — edges route to rel_Folder_Folder.csv (#2203 U2). Deliberately NOT + // in insertion order (c, a, b). const NODES: TestNodeInput[] = [ - { id: 'folder:c', label: 'Folder', name: 'c', filePath: 'c' }, - { id: 'folder:a', label: 'Folder', name: 'a', filePath: 'a' }, - { id: 'folder:b', label: 'Folder', name: 'b', filePath: 'b' }, + { id: 'Folder:c', label: 'Folder', name: 'c', filePath: 'c' }, + { id: 'Folder:a', label: 'Folder', name: 'a', filePath: 'a' }, + { id: 'Folder:b', label: 'Folder', name: 'b', filePath: 'b' }, ]; const RELS: TestRelInput[] = [ - { sourceId: 'folder:c', targetId: 'folder:a', type: 'CONTAINS' }, - { sourceId: 'folder:a', targetId: 'folder:b', type: 'CONTAINS' }, - { sourceId: 'folder:b', targetId: 'folder:c', type: 'CONTAINS' }, + { sourceId: 'Folder:c', targetId: 'Folder:a', type: 'CONTAINS' }, + { sourceId: 'Folder:a', targetId: 'Folder:b', type: 'CONTAINS' }, + { sourceId: 'Folder:b', targetId: 'Folder:c', type: 'CONTAINS' }, ]; const dataRows = (csv: string): string[] => csv @@ -270,7 +326,7 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => { const folderIds = folderCsv ? dataRows(await fs.readFile(folderCsv.csvPath, 'utf-8')).map(firstCol) : []; - const relRows = dataRows(await fs.readFile(result.relCsvPath, 'utf-8')); + const relRows = await readAllRelRows(result.relsByPair); return { folderIds, relRows }; } finally { delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT; @@ -307,3 +363,205 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => { expect([...onFwd.relRows].sort()).toEqual([...offFwd.relRows].sort()); }); }); + +/** + * #2203 U2 byte-identity: for all quote-free ids the direct per-pair emit must + * produce per-pair files byte-for-byte identical to the legacy + * splitRelCsvByLabelPair oracle run over an equivalent monolithic relations.csv + * from the same graph. This is the load-bearing guard for "byte-identical graph + * content" (issue acceptance). The ONE intentional divergence — ids containing a + * double-quote, where the router (raw-id label) is more correct than the oracle + * (regex over the escaped row) — is asserted explicitly in its own test below. + */ +describe('streamAllCSVsToDisk — direct per-pair emit matches the split oracle', () => { + // The oracle always emits in graph.iterRelationships() (unsorted) order; the + // production path honours GITNEXUS_SORT_GRAPH_OUTPUT. Clear it so a value + // leaked from a prior test can't desync the two and produce a spurious diff. + beforeEach(() => { + delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT; + }); + + it('produces byte-identical per-pair files + identical skip/total accounting', async () => { + // Multiple valid pairs, getNodeLabel special prefixes (comm_ AND proc_), and + // one invalid-label edge that BOTH paths must skip identically. + const graph = buildTestGraph( + [ + { id: 'File:a.ts', label: 'File', name: 'a.ts', filePath: 'a.ts' }, + { id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' }, + { id: 'Function:a.ts:g:5', label: 'Function', name: 'g', filePath: 'a.ts' }, + { id: 'comm_1', label: 'Community' as never, name: 'c1', filePath: '' }, + { id: 'comm_2', label: 'Community' as never, name: 'c2', filePath: '' }, + { id: 'proc_1', label: 'Process' as never, name: 'p1', filePath: '' }, + { id: 'proc_2', label: 'Process' as never, name: 'p2', filePath: '' }, + ], + [ + { sourceId: 'File:a.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' }, + { sourceId: 'File:a.ts', targetId: 'Function:a.ts:g:5', type: 'CONTAINS' }, + { sourceId: 'Function:a.ts:f:1', targetId: 'Function:a.ts:g:5', type: 'CALLS' }, + { sourceId: 'comm_1', targetId: 'comm_2', type: 'CONTAINS' }, + // proc_ prefix → Process label (getNodeLabel special case). + { sourceId: 'proc_1', targetId: 'proc_2', type: 'CONTAINS' }, + // Invalid FROM label ('Bogus' ∉ NODE_TABLES) — skipped by both paths. + { sourceId: 'Bogus:x', targetId: 'File:a.ts', type: 'CONTAINS' }, + // Invalid TO label — exercises the OTHER branch of the skip condition. + { sourceId: 'File:a.ts', targetId: 'Bogus:y', type: 'CONTAINS' }, + ], + ); + + const directDir = path.join(csvDir, 'diff-direct'); + const oracleDir = path.join(csvDir, 'diff-oracle'); + await fs.mkdir(oracleDir, { recursive: true }); + + // Direct emit (production path). + const direct = await streamAllCSVsToDisk(graph, repoDir, directDir); + + // Oracle: build the monolithic relations.csv this graph would have produced + // (same insertion order, same row bytes via buildRelRow), then split it. + const relCsv = path.join(oracleDir, 'relations.csv'); + const lines = [REL_CSV_HEADER]; + for (const rel of graph.iterRelationships()) lines.push(buildRelRow(rel)); + await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8'); + + const split = await splitRelCsvByLabelPair( + relCsv, + oracleDir, + new Set(NODE_TABLES), + getNodeLabel, + ); + await Promise.all( + Array.from(split.pairWriteStreams.values()).map(async (ws) => { + ws.end(); + await finished(ws); + }), + ); + + // Identical accounting. + expect(direct.totalValidRels).toBe(split.totalValidRels); + expect(direct.totalValidRels).toBe(5); + expect(direct.skippedRels).toBe(split.skippedRels); + expect(direct.skippedRels).toBe(2); // invalid-FROM + invalid-TO, both skipped + expect(direct.relHeader).toBe(split.relHeader); + + // Identical pair set. + expect([...direct.relsByPair.keys()].sort()).toEqual([...split.relsByPairMeta.keys()].sort()); + + // Byte-identical per-pair file contents. + for (const key of direct.relsByPair.keys()) { + const directContent = await fs.readFile(direct.relsByPair.get(key)!.csvPath, 'utf-8'); + const oracleContent = await fs.readFile(split.relsByPairMeta.get(key)!.csvPath, 'utf-8'); + expect(directContent, `pair ${key}`).toBe(oracleContent); + } + }); + + it('quote-in-id edge: router routes it (raw-id label) while the oracle drops it — intended divergence', async () => { + // A node id with an embedded double-quote (legal in a POSIX filePath). The + // router derives the label from the RAW id (`File`), so it routes the edge; + // the oracle re-derives the label via /"([^"]*)","([^"]*)"/ over the ESCAPED + // row (`"File:a""b.ts",...`), mis-reads the field, and drops it. This locks + // the intended divergence so a future change can't silently revert the + // router to the buggy regex semantics. + const graph = buildTestGraph( + [ + { id: 'File:clean.ts', label: 'File', name: 'clean.ts', filePath: 'clean.ts' }, + { id: 'File:a"b.ts', label: 'File', name: 'a"b.ts', filePath: 'a"b.ts' }, + { id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' }, + ], + [ + { sourceId: 'File:clean.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' }, + { sourceId: 'File:a"b.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' }, + ], + ); + + const directDir = path.join(csvDir, 'qd-direct'); + const oracleDir = path.join(csvDir, 'qd-oracle'); + await fs.mkdir(oracleDir, { recursive: true }); + + const direct = await streamAllCSVsToDisk(graph, repoDir, directDir); + + const relCsv = path.join(oracleDir, 'relations.csv'); + const lines = [REL_CSV_HEADER]; + for (const rel of graph.iterRelationships()) lines.push(buildRelRow(rel)); + await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8'); + const split = await splitRelCsvByLabelPair( + relCsv, + oracleDir, + new Set(NODE_TABLES), + getNodeLabel, + ); + await Promise.all( + Array.from(split.pairWriteStreams.values()).map(async (ws) => { + ws.end(); + await finished(ws); + }), + ); + + // Router routes BOTH edges — the raw-id label `File` is valid for both. + expect(direct.totalValidRels).toBe(2); + expect(direct.skippedRels).toBe(0); + expect(direct.relsByPair.get('File|Function')!.rows).toBe(2); + + // Oracle DIVERGES: its regex mis-reads the quote-in-id row and drops that + // edge, so it routes strictly fewer edges. Asserted robustly — we do NOT + // pin the oracle's exact mis-derived label. + expect(split.totalValidRels).toBeLessThan(direct.totalValidRels); + expect(split.skippedRels).toBeGreaterThan(direct.skippedRels); + }); + + it('sorted path (GITNEXUS_SORT_GRAPH_OUTPUT=1): per-pair files byte-identical to the oracle', async () => { + // The earlier differential test covers the default (insertion-order) path. + // Here the sorted emit path must also match the oracle — fed the SAME + // id-sorted order orderedRelationships() uses (sort by rel.id). + process.env.GITNEXUS_SORT_GRAPH_OUTPUT = '1'; + try { + const graph = buildTestGraph( + [ + { id: 'File:a.ts', label: 'File', name: 'a.ts', filePath: 'a.ts' }, + { id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' }, + { id: 'Function:a.ts:g:5', label: 'Function', name: 'g', filePath: 'a.ts' }, + ], + // Deliberately NOT in id-sorted order so the sort actually reorders rows. + [ + { sourceId: 'Function:a.ts:f:1', targetId: 'Function:a.ts:g:5', type: 'CALLS' }, + { sourceId: 'File:a.ts', targetId: 'Function:a.ts:g:5', type: 'CONTAINS' }, + { sourceId: 'File:a.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' }, + ], + ); + + const directDir = path.join(csvDir, 'sorted-direct'); + const oracleDir = path.join(csvDir, 'sorted-oracle'); + await fs.mkdir(oracleDir, { recursive: true }); + + const direct = await streamAllCSVsToDisk(graph, repoDir, directDir); + + // Oracle fed the same id-sorted order the sorted emit produces. + const sortedRels = [...graph.iterRelationships()].sort((a, b) => + a.id < b.id ? -1 : a.id > b.id ? 1 : 0, + ); + const relCsv = path.join(oracleDir, 'relations.csv'); + const lines = [REL_CSV_HEADER]; + for (const rel of sortedRels) lines.push(buildRelRow(rel)); + await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8'); + const split = await splitRelCsvByLabelPair( + relCsv, + oracleDir, + new Set(NODE_TABLES), + getNodeLabel, + ); + await Promise.all( + Array.from(split.pairWriteStreams.values()).map(async (ws) => { + ws.end(); + await finished(ws); + }), + ); + + expect([...direct.relsByPair.keys()].sort()).toEqual([...split.relsByPairMeta.keys()].sort()); + for (const key of direct.relsByPair.keys()) { + const directContent = await fs.readFile(direct.relsByPair.get(key)!.csvPath, 'utf-8'); + const oracleContent = await fs.readFile(split.relsByPairMeta.get(key)!.csvPath, 'utf-8'); + expect(directContent, `pair ${key} (sorted)`).toBe(oracleContent); + } + } finally { + delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT; + } + }); +}); diff --git a/gitnexus/test/integration/lbug-load-prof.test.ts b/gitnexus/test/integration/lbug-load-prof.test.ts new file mode 100644 index 000000000..f83cd75c2 --- /dev/null +++ b/gitnexus/test/integration/lbug-load-prof.test.ts @@ -0,0 +1,141 @@ +/** + * Integration test: PROF_LBUG_LOAD persistence-path profiling (#2203 U1). + * + * loadGraphToLbug is un-timed in production today; the analyze "emit" number + * is the scope-resolution emit bucket, not this CSV→COPY persistence path. + * U1 adds a zero-cost-when-off per-stage breakdown gated by PROF_LBUG_LOAD=1, + * mirroring the PROF_SCOPE_RESOLUTION pattern. These tests assert the gate: + * - flag off → no `[lbug-load prof]` line is logged, behaviour unchanged + * - flag on → exactly one summary line with every stage key + node/rel counts + * + * Needs a real LadybugDB connection (initLbug), so it lives under integration. + * Logger assertions use `_captureLogger()` — the exported `logger` is a Proxy + * over a lazily-built pino instance and is not directly spy-able. + */ +import { describe, it, expect, beforeAll, beforeEach, afterAll, afterEach } from 'vitest'; +import fs from 'fs/promises'; +import path from 'path'; +import os from 'os'; +import { buildTestGraph } from '../helpers/test-graph.js'; +import { _captureLogger, type LoggerCapture } from '../../src/core/logger.js'; + +let tmpBase: string; +let storagePath: string; +let dbPath: string; +let cap: LoggerCapture; + +const PROF_LINE = '[lbug-load prof]'; + +const profLines = (): string[] => + cap + .records() + .map((r) => (typeof r.msg === 'string' ? r.msg : '')) + .filter((msg) => msg.includes(PROF_LINE)); + +beforeAll(async () => { + tmpBase = path.join(os.tmpdir(), `gitnexus-lbug-prof-${Date.now()}-${process.pid}`); + storagePath = path.join(tmpBase, '.gitnexus'); + dbPath = path.join(storagePath, 'lbug'); + await fs.mkdir(dbPath, { recursive: true }); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.initLbug(dbPath); +}); + +beforeEach(() => { + cap = _captureLogger(); +}); + +afterEach(() => { + cap.restore(); + delete process.env.PROF_LBUG_LOAD; +}); + +afterAll(async () => { + try { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.closeLbug(); + } catch { + /* may not have opened */ + } + try { + await fs.rm(tmpBase, { recursive: true, force: true }); + } catch { + /* best-effort */ + } +}); + +describe('PROF_LBUG_LOAD persistence-path profiling (#2203 U1)', () => { + it('does NOT log a prof summary when the flag is unset', async () => { + delete process.env.PROF_LBUG_LOAD; + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + const graph = buildTestGraph( + [ + { id: 'File:src/off.ts', label: 'File', name: 'off.ts', filePath: 'src/off.ts' }, + { + id: 'Function:src/off.ts:offFn:1', + label: 'Function', + name: 'offFn', + filePath: 'src/off.ts', + startLine: 1, + endLine: 2, + }, + ], + [{ sourceId: 'File:src/off.ts', targetId: 'Function:src/off.ts:offFn:1', type: 'DEFINES' }], + ); + + const result = await adapter.loadGraphToLbug(graph, tmpBase, storagePath); + + expect(result.success).toBe(true); + expect(profLines()).toHaveLength(0); + }); + + it('logs exactly one summary line with all stage keys + counts when the flag is set', async () => { + process.env.PROF_LBUG_LOAD = '1'; + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + // Distinct ids from the flag-off graph so the COPY does not hit a + // PK-dup IGNORE_ERRORS retry on the shared singleton connection. + const graph = buildTestGraph( + [ + { id: 'File:src/on.ts', label: 'File', name: 'on.ts', filePath: 'src/on.ts' }, + { + id: 'Function:src/on.ts:onFn:1', + label: 'Function', + name: 'onFn', + filePath: 'src/on.ts', + startLine: 1, + endLine: 2, + }, + { + id: 'Class:src/on.ts:OnClass:5', + label: 'Class', + name: 'OnClass', + filePath: 'src/on.ts', + startLine: 5, + endLine: 8, + }, + ], + [ + { sourceId: 'File:src/on.ts', targetId: 'Function:src/on.ts:onFn:1', type: 'DEFINES' }, + { sourceId: 'File:src/on.ts', targetId: 'Class:src/on.ts:OnClass:5', type: 'DEFINES' }, + ], + ); + + const result = await adapter.loadGraphToLbug(graph, tmpBase, storagePath); + expect(result.success).toBe(true); + + const lines = profLines(); + expect(lines).toHaveLength(1); + + const line = lines[0]; + // Relationships are routed to per-pair files during csv-emit (#2203 U2), + // so there is no separate rel-split stage. + for (const key of ['csv-emit=', 'copy-nodes=', 'copy-rels=', 'fallback=', 'total=']) { + expect(line).toContain(key); + } + // 3 node rows (File, Function, Class), 2 valid rels emitted. + expect(line).toContain('(3 nodes, 2 rels)'); + }); +}); diff --git a/gitnexus/test/unit/rel-pair-routing.test.ts b/gitnexus/test/unit/rel-pair-routing.test.ts new file mode 100644 index 000000000..8355014d7 --- /dev/null +++ b/gitnexus/test/unit/rel-pair-routing.test.ts @@ -0,0 +1,176 @@ +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { EventEmitter } from 'events'; +import fs from 'fs'; +import path from 'path'; +import os from 'os'; +import { RelPairRouter, getNodeLabel } from '../../src/core/lbug/rel-pair-routing.js'; + +/** + * Unit tests for RelPairRouter (#2203 U2) — the production per-pair emit path. + * + * Mirrors test/unit/rel-csv-split.test.ts: drives the router with an injected + * mock WriteStream factory so the error, backpressure, and teardown paths are + * exercised without LadybugDB or real disk streams. These paths are otherwise + * unreachable in the integration suite (which only hits the no-backpressure + * happy path), so this is the coverage for the router's failure modes. + */ + +// Controllable backpressure + error injection (same shape as the split oracle's mock). +class MockWriteStream extends EventEmitter { + public chunks: string[] = []; + public destroyed = false; + public ended = false; + public blocked = false; + public maxDrainListenersSeen = 0; + // State flags + events so `stream/promises.finished(ws)` (used by the + // router's close()) resolves against this mock instead of hanging. + public writable = true; + public writableEnded = false; + public writableFinished = false; + + write(chunk: string): boolean { + this.chunks.push(chunk); + const count = this.listenerCount('drain'); + if (count > this.maxDrainListenersSeen) this.maxDrainListenersSeen = count; + return !this.blocked; + } + + end(cb?: (err?: Error) => void): this { + this.ended = true; + this.writableEnded = true; + this.writableFinished = true; + this.writable = false; + if (cb) cb(); + queueMicrotask(() => { + this.emit('finish'); + this.emit('close'); + }); + return this; + } + + destroy(): this { + this.destroyed = true; + return this; + } + + unblock(): void { + this.blocked = false; + this.emit('drain'); + } + + triggerError(err: Error): void { + this.emit('error', err); + } +} + +const HEADER = '"from","to","type","confidence","reason","step"'; +const VALID = new Set(['File', 'Function', 'Community', 'Process']); + +const row = (from: string, to: string, type = 'CALLS'): string => + `"${from}","${to}","${type}",1.0,"auto",0`; + +function mockFactory(streams: MockWriteStream[], opts?: { blocked?: boolean }) { + return (() => { + const ws = new MockWriteStream(); + if (opts?.blocked) ws.blocked = true; + streams.push(ws); + return ws; + }) as unknown as (filePath: string) => import('fs').WriteStream; +} + +let tmpDir: string; + +beforeEach(() => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'rel-pair-routing-test-')); +}); + +afterEach(() => { + fs.rmSync(tmpDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 50 }); +}); + +describe('getNodeLabel', () => { + it('maps comm_/proc_ prefixes and otherwise splits on the first colon', () => { + expect(getNodeLabel('comm_42')).toBe('Community'); + expect(getNodeLabel('proc_7')).toBe('Process'); + expect(getNodeLabel('Function:src/a.ts:f:1')).toBe('Function'); + expect(getNodeLabel('File:src/a.ts')).toBe('File'); + }); +}); + +describe('RelPairRouter', () => { + it('routes valid edges to per-pair files (header first) and skips invalid-label edges', async () => { + const streams: MockWriteStream[] = []; + const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams)); + + const route = async (from: string, to: string) => { + const p = router.route(from, to, row(from, to)); + if (p) await p; + }; + await route('File:a', 'Function:a:f:1'); + await route('File:a', 'Function:a:g:2'); // same pair + await route('Function:a:f:1', 'Function:a:g:2'); // different pair + await route('Bogus:x', 'File:a'); // invalid FROM label → skipped + await route('File:a', 'Bogus:y'); // invalid TO label → skipped (other branch) + await router.close(); + + expect(router.skipped).toBe(2); + expect(router.total).toBe(3); + expect([...router.byPair.keys()].sort()).toEqual(['File|Function', 'Function|Function']); + expect(router.byPair.get('File|Function')!.rows).toBe(2); + // Header is the first chunk written to each pair stream. + expect(streams[0].chunks[0]).toBe(HEADER + '\n'); + expect(streams.every((s) => s.ended)).toBe(true); + }); + + it('returns a drain promise under backpressure and completes once unblocked', async () => { + const streams: MockWriteStream[] = []; + const router = new RelPairRouter( + tmpDir, + HEADER, + VALID, + mockFactory(streams, { blocked: true }), + ); + + const pending = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1')); + expect(pending).toBeInstanceOf(Promise); // header write hit backpressure + streams[0].unblock(); + await pending; + + expect(streams[0].maxDrainListenersSeen).toBeLessThanOrEqual(1); + expect(streams[0].chunks[0]).toBe(HEADER + '\n'); + expect(router.total).toBe(1); + }); + + it('on a stream error: route() throws the real error, lastError exposes it, close() rejects + destroys', async () => { + const streams: MockWriteStream[] = []; + const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams)); + + const first = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1')); + if (first) await first; + + const err = new Error('EMFILE: too many open files'); + streams[0].triggerError(err); + + // The next route surfaces the REAL error, not a generic AbortError. + expect(() => router.route('File:a', 'Function:a:g:2', row('File:a', 'Function:a:g:2'))).toThrow( + 'EMFILE', + ); + expect(router.lastError).toBe(err); + await expect(router.close()).rejects.toThrow('EMFILE'); + expect(streams[0].destroyed).toBe(true); + }); + + it('destroy() tears down every open pair stream', async () => { + const streams: MockWriteStream[] = []; + const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams)); + + const a = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1')); + if (a) await a; + const b = router.route('Community:1', 'Community:2', row('Community:1', 'Community:2')); + if (b) await b; + + router.destroy(); + expect(streams.length).toBe(2); + expect(streams.every((s) => s.destroyed)).toBe(true); + }); +}); From 2d389a7d1b4fa3436361236df02745c3c3ba2f31 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 15 Jun 2026 21:59:35 +0100 Subject: [PATCH 05/26] chore(deps-dev): bump js-yaml (#2217) Bumps the npm_and_yarn group with 1 update in the / directory: [js-yaml](https://github.com/nodeca/js-yaml). Updates `js-yaml` from 4.1.1 to 4.2.0 - [Changelog](https://github.com/nodeca/js-yaml/blob/master/CHANGELOG.md) - [Commits](https://github.com/nodeca/js-yaml/commits) --- updated-dependencies: - dependency-name: js-yaml dependency-version: 4.2.0 dependency-type: indirect dependency-group: npm_and_yarn ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- package-lock.json | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/package-lock.json b/package-lock.json index 65c0870ed..0969cc7a7 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1868,10 +1868,20 @@ "license": "MIT" }, "node_modules/js-yaml": { - "version": "4.1.1", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.1.1.tgz", - "integrity": "sha512-qQKT4zQxXl8lLwBtHMWwaTcGfFOZviOJet3Oy/xmGk2gZH677CJM9EvtfdSkgWcATZhj/55JZ0rmy3myCT5lsA==", + "version": "4.2.0", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.2.0.tgz", + "integrity": "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw==", "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/puzrin" + }, + { + "type": "github", + "url": "https://github.com/sponsors/nodeca" + } + ], "license": "MIT", "dependencies": { "argparse": "^2.0.1" From fbbda5a19b33dacb8267a82e60ffafd2c174dfdc Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 15 Jun 2026 21:59:48 +0100 Subject: [PATCH 06/26] chore(deps)(deps): bump tar from 7.5.13 to 7.5.16 in /gitnexus (#2218) Bumps [tar](https://github.com/isaacs/node-tar) from 7.5.13 to 7.5.16. - [Release notes](https://github.com/isaacs/node-tar/releases) - [Changelog](https://github.com/isaacs/node-tar/blob/main/CHANGELOG.md) - [Commits](https://github.com/isaacs/node-tar/compare/v7.5.13...v7.5.16) --- updated-dependencies: - dependency-name: tar dependency-version: 7.5.16 dependency-type: indirect ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 991729ee6..1a5d5d467 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -4917,9 +4917,9 @@ } }, "node_modules/tar": { - "version": "7.5.13", - "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.13.tgz", - "integrity": "sha512-tOG/7GyXpFevhXVh8jOPJrmtRpOTsYqUIkVdVooZYJS/z8WhfQUX8RJILmeuJNinGAMSu1veBr4asSHFt5/hng==", + "version": "7.5.16", + "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz", + "integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==", "license": "BlueOak-1.0.0", "dependencies": { "@isaacs/fs-minipass": "^4.0.0", From 526194bf426ee7bdd3928ac2c840f7533d67c7ea Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 15 Jun 2026 22:00:14 +0100 Subject: [PATCH 07/26] chore(deps)(deps): bump protobufjs from 7.5.8 to 7.6.4 in /gitnexus (#2219) Bumps [protobufjs](https://github.com/protobufjs/protobuf.js) from 7.5.8 to 7.6.4. - [Release notes](https://github.com/protobufjs/protobuf.js/releases) - [Changelog](https://github.com/protobufjs/protobuf.js/blob/protobufjs-v7.6.4/CHANGELOG.md) - [Commits](https://github.com/protobufjs/protobuf.js/compare/protobufjs-v7.5.8...protobufjs-v7.6.4) --- updated-dependencies: - dependency-name: protobufjs dependency-version: 7.6.4 dependency-type: indirect ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 34 +++++++++++++--------------------- 1 file changed, 13 insertions(+), 21 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 1a5d5d467..5bf81d34a 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -1340,19 +1340,18 @@ "license": "BSD-3-Clause" }, "node_modules/@protobufjs/eventemitter": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.0.tgz", - "integrity": "sha512-j9ednRT81vYJ9OfVuXG6ERSTdEL1xVsNgqpkxMsbIabzSo3goCjDIveeGv5d03om39ML71RdmrGNjG5SReBP/Q==", + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.1.tgz", + "integrity": "sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg==", "license": "BSD-3-Clause" }, "node_modules/@protobufjs/fetch": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.0.tgz", - "integrity": "sha512-lljVXpqXebpsijW71PZaCYeIcE5on1w5DlQy5WH6GLbFryLUrBD4932W/E2BSpfRJWseIL4v/KPgBFxDOIdKpQ==", + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.1.tgz", + "integrity": "sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw==", "license": "BSD-3-Clause", "dependencies": { - "@protobufjs/aspromise": "^1.1.1", - "@protobufjs/inquire": "^1.1.0" + "@protobufjs/aspromise": "^1.1.1" } }, "node_modules/@protobufjs/float": { @@ -1361,12 +1360,6 @@ "integrity": "sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ==", "license": "BSD-3-Clause" }, - "node_modules/@protobufjs/inquire": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@protobufjs/inquire/-/inquire-1.1.1.tgz", - "integrity": "sha512-mnzgDV26ueAvk7rsbt9L7bE0SuAoqyuys/sMMrmVcN5x9VsxpcG3rqAUSgDyLp0UZlmNfIbQ4fHfCtreVBk8Ew==", - "license": "BSD-3-Clause" - }, "node_modules/@protobufjs/path": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/@protobufjs/path/-/path-1.1.2.tgz", @@ -4351,24 +4344,23 @@ "license": "MIT" }, "node_modules/protobufjs": { - "version": "7.5.8", - "resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.5.8.tgz", - "integrity": "sha512-dvpCIeLPbXZS/Ete7yLaO7RenOdken2NHKykBXbsaGxZT0UTltcarBciw+A78SRQs9iMAAVpsYA+l8b1hTePIA==", + "version": "7.6.4", + "resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.6.4.tgz", + "integrity": "sha512-RJJPTTpvFfHcWLkIa2JFWK4XvtSzS0yEWDmunqHXli1h3JlkbcQZXDZdcWxv+JK3Xsl5/UFDPZ0iGm7DAengYw==", "hasInstallScript": true, "license": "BSD-3-Clause", "dependencies": { "@protobufjs/aspromise": "^1.1.2", "@protobufjs/base64": "^1.1.2", "@protobufjs/codegen": "^2.0.5", - "@protobufjs/eventemitter": "^1.1.0", - "@protobufjs/fetch": "^1.1.0", + "@protobufjs/eventemitter": "^1.1.1", + "@protobufjs/fetch": "^1.1.1", "@protobufjs/float": "^1.0.2", - "@protobufjs/inquire": "^1.1.1", "@protobufjs/path": "^1.1.2", "@protobufjs/pool": "^1.1.0", "@protobufjs/utf8": "^1.1.1", "@types/node": ">=13.7.0", - "long": "^5.0.0" + "long": "^5.3.2" }, "engines": { "node": ">=12.0.0" From 067c73b6b2f214b21e5968535476c81e1717c0b6 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 15 Jun 2026 22:01:02 +0100 Subject: [PATCH 08/26] chore(deps)(deps-dev): bump @types/node in /gitnexus (#2222) Bumps [@types/node](https://github.com/DefinitelyTyped/DefinitelyTyped/tree/HEAD/types/node) from 25.9.2 to 25.9.3. - [Release notes](https://github.com/DefinitelyTyped/DefinitelyTyped/releases) - [Commits](https://github.com/DefinitelyTyped/DefinitelyTyped/commits/HEAD/types/node) --- updated-dependencies: - dependency-name: "@types/node" dependency-version: 25.9.3 dependency-type: direct:development update-type: version-update:semver-patch ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 5bf81d34a..3a93b9339 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -1815,9 +1815,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "25.9.2", - "resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.2.tgz", - "integrity": "sha512-G05zqtJhcDLb8uslf5EjCxXg9G1KQxiV8OS0R26IC//Eoyitzqe8z37I7cqvnZlrlSfgocQRfSn/AHBZJJFyGw==", + "version": "25.9.3", + "resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.3.tgz", + "integrity": "sha512-603BddQMv3pUcr4U2dhujk83N2tTDVr/34wII2B6bJy6g+8WD6yUb11jszNs0gdi4PesVWl7ABt8nYMVpnLUcg==", "license": "MIT", "dependencies": { "undici-types": ">=7.24.0 <7.24.7" From 3c82361b66b67832e9b612e5aca1c40776e15d33 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 16 Jun 2026 05:04:10 +0100 Subject: [PATCH 09/26] perf(cfg): streaming/chunked PDG graph emit for full-kernel-scale repos (#2202) (#2216) --- .github/workflows/ci-tests.yml | 9 + .../emit-persistence/baselines-streaming.json | 4 + .../emit-persistence/measure-streaming.mjs | 199 +++++++ gitnexus/src/core/ingestion/pipeline.ts | 26 +- .../scope-resolution/pipeline/phase.ts | 523 ++++++++++-------- .../scope-resolution/pipeline/run.ts | 46 +- gitnexus/src/core/ingestion/utils/env.ts | 12 + gitnexus/src/core/lbug/csv-generator.ts | 30 +- gitnexus/src/core/lbug/lbug-adapter.ts | 54 +- gitnexus/src/core/lbug/lbug-config.ts | 30 + gitnexus/src/core/lbug/pdg-emit-sink.ts | 395 +++++++++++++ gitnexus/src/core/run-analyze.ts | 81 ++- gitnexus/src/types/pipeline.ts | 9 + .../cfg/fixtures/vue-ts-pdg/app.vue | 28 + .../cfg/fixtures/vue-ts-pdg/shared.ts | 44 ++ .../cfg/pipeline-pdg-streaming.test.ts | 167 ++++++ .../pdg-emit-streaming-roundtrip.test.ts | 181 ++++++ .../test/unit/lbug-native-safe-path.test.ts | 56 +- gitnexus/test/unit/lbug/pdg-emit-sink.test.ts | 331 +++++++++++ .../test/unit/stream-pdg-emit-config.test.ts | 99 ++++ 20 files changed, 2064 insertions(+), 260 deletions(-) create mode 100644 gitnexus/bench/emit-persistence/baselines-streaming.json create mode 100644 gitnexus/bench/emit-persistence/measure-streaming.mjs create mode 100644 gitnexus/src/core/lbug/pdg-emit-sink.ts create mode 100644 gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue create mode 100644 gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts create mode 100644 gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts create mode 100644 gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts create mode 100644 gitnexus/test/unit/lbug/pdg-emit-sink.test.ts create mode 100644 gitnexus/test/unit/stream-pdg-emit-config.test.ts diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index a851696fb..4b0548e4f 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -285,6 +285,15 @@ jobs: run: node --import tsx bench/emit-persistence/measure.mjs --check working-directory: gitnexus + - name: Streaming PDG-emit byte-identity / bounded-RSS guards (#2202) + # Build-free: asserts the streaming PdgEmitSink emits a CSV row SET + # byte-identical to the whole-graph streamAllCSVsToDisk emit, AND that + # the in-memory graph retains zero BasicBlock nodes (the O(chunk) peak-RSS + # bound that unblocks full-kernel-scale repos). Fails on fingerprint drift + # or any resident BasicBlock. + run: node --import tsx bench/emit-persistence/measure-streaming.mjs --check + working-directory: gitnexus + - name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial) env: GITNEXUS_BENCH: '1' diff --git a/gitnexus/bench/emit-persistence/baselines-streaming.json b/gitnexus/bench/emit-persistence/baselines-streaming.json new file mode 100644 index 000000000..62e1c0725 --- /dev/null +++ b/gitnexus/bench/emit-persistence/baselines-streaming.json @@ -0,0 +1,4 @@ +{ + "fingerprint": "386f432c74f4992455055d8891dbe6c873ea95afa60ef4be023a21aed7b4bcb1", + "_note": "Byte-identity + bounded-retention gate for streaming/chunked PDG emit (#2202). fingerprint = sha256 of the sorted, header-stripped BasicBlock + PDG-edge data rows of the canonical synthetic set. --check also asserts the streamed PdgEmitSink output is byte-identical to the whole-graph streamAllCSVsToDisk emit (byte_identical_nodes/edges) and that the in-memory graph retains 0 BasicBlocks (resident_basic_blocks === 0, the O(chunk) RSS bound). Regenerate via `node --import tsx bench/emit-persistence/measure-streaming.mjs`." +} diff --git a/gitnexus/bench/emit-persistence/measure-streaming.mjs b/gitnexus/bench/emit-persistence/measure-streaming.mjs new file mode 100644 index 000000000..683b86888 --- /dev/null +++ b/gitnexus/bench/emit-persistence/measure-streaming.mjs @@ -0,0 +1,199 @@ +/** + * Build-free byte-identity + bounded-retention bench for streaming/chunked PDG + * graph emit (issue #2202). + * + * Proves the two acceptance criteria at scale, without a DB connection: + * 1. BYTE-IDENTITY (R2): emitting a BasicBlock + intra-file PDG-edge set via + * the streaming `PdgEmitSink` produces the IDENTICAL CSV data-row set as + * the whole-graph `streamAllCSVsToDisk` path. Compared per file + * (basicblock.csv, rel_BasicBlock_BasicBlock.csv) over header-stripped, + * sorted lines so it is a pure function of the emitted row SET. + * 2. BOUNDED RETENTION (R1): with streaming on, the in-memory graph holds + * ZERO BasicBlock nodes regardless of how many are emitted — the PDG layer + * never accumulates in process memory (peak RSS O(chunk), not O(graph)). + * + * Build-free: imports the `.ts` hotpaths through tsx + * (`node --import tsx bench/emit-persistence/measure-streaming.mjs`). + * + * Without args: prints one JSON object. With `--check`: asserts byte-identity, + * retention, and fingerprint == the committed baseline; exits non-zero on any + * failure. + */ +import fs from 'node:fs'; +import fsp from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { fileURLToPath } from 'node:url'; + +import { createKnowledgeGraph } from '../../src/core/graph/graph.ts'; +import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.ts'; +import { PdgEmitSink } from '../../src/core/lbug/pdg-emit-sink.ts'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines-streaming.json'); + +// PDG edge types streamed per file (all intra-block BasicBlock→BasicBlock). +const PDG_TYPES = ['CFG', 'REACHING_DEF', 'CDG', 'POST_DOMINATE', 'TAINTED', 'SANITIZES']; + +const FUNCS = 1200; // functions +const BLOCKS = 6; // basic blocks per function ⇒ FUNCS*BLOCKS BasicBlocks total +const CHUNK_ROWS = 64; // tiny streamed buffer to exercise frequent flushing + +/** + * Build the canonical PDG node/edge SET: `FUNCS` functions each with `BLOCKS` + * BasicBlocks and a chain of intra-function PDG edges. Returns the structural + * nodes (File/Function) separately from the BasicBlock + PDG-edge layer so the + * streamed path can route them to different sinks. + */ +function buildSet() { + const structuralNodes = []; + const structuralRels = []; + const bbNodes = []; + const pdgEdges = []; + for (let f = 0; f < FUNCS; f++) { + const fp = `src/m${f % 50}.ts`; + const fnId = `Function:${fp}:fn${f}:1`; + structuralNodes.push({ + id: fnId, + label: 'Function', + properties: { name: `fn${f}`, filePath: fp, startLine: 1, endLine: 99 }, + }); + for (let b = 0; b < BLOCKS; b++) { + bbNodes.push({ + id: `BasicBlock:${fp}:1:0:${f}_${b}`, + label: 'BasicBlock', + properties: { + name: '', + filePath: fp, + startLine: b * 3, + endLine: b * 3 + 2, + text: `f${f}b${b}`, + }, + }); + } + for (let b = 0; b < BLOCKS - 1; b++) { + const from = `BasicBlock:${fp}:1:0:${f}_${b}`; + const to = `BasicBlock:${fp}:1:0:${f}_${b + 1}`; + for (const type of PDG_TYPES) { + pdgEdges.push({ + id: `${type}:${f}:${b}`, + sourceId: from, + targetId: to, + type, + confidence: 1, + reason: type === 'REACHING_DEF' ? `v${b}` : type === 'CDG' ? 'T' : '', + }); + } + } + } + // A few File nodes so the structural emit produces a realistic multi-table mix. + for (let m = 0; m < 50; m++) { + structuralNodes.push({ + id: `File:src/m${m}.ts`, + label: 'File', + properties: { name: `m${m}.ts`, filePath: `src/m${m}.ts` }, + }); + } + return { structuralNodes, structuralRels, bbNodes, pdgEdges }; +} + +/** Header-stripped, sorted, non-empty data rows of one CSV file (or [] if absent). */ +async function dataRows(csvPath) { + let text; + try { + text = await fsp.readFile(csvPath, 'utf8'); + } catch { + return []; + } + const lines = text.split('\n').filter((l) => l.length > 0); + return lines.slice(1).sort(); // drop the header line +} + +const sha = (rows) => crypto.createHash('sha256').update(rows.join('\n')).digest('hex'); + +async function measure() { + // mkdtemp (unpredictable, unique) rather than a predictable pid-based tmp path. + const tmpRoot = await fsp.mkdtemp(path.join(os.tmpdir(), 'gitnexus-stream-bench-')); + try { + const { structuralNodes, structuralRels, bbNodes, pdgEdges } = buildSet(); + + // ── whole-graph path ───────────────────────────────────────────────── + const wholeGraph = createKnowledgeGraph(); + for (const n of structuralNodes) wholeGraph.addNode(n); + for (const n of bbNodes) wholeGraph.addNode(n); + for (const r of structuralRels) wholeGraph.addRelationship(r); + for (const e of pdgEdges) wholeGraph.addRelationship(e); + const wholeDir = path.join(tmpRoot, 'whole'); + await streamAllCSVsToDisk(wholeGraph, path.join(tmpRoot, 'no-repo'), wholeDir); + + // ── streamed path ──────────────────────────────────────────────────── + const realGraph = createKnowledgeGraph(); + const sink = new PdgEmitSink(realGraph, path.join(tmpRoot, 'pdg-csv'), CHUNK_ROWS); + for (const n of structuralNodes) realGraph.addNode(n); // structural → real graph + for (const r of structuralRels) realGraph.addRelationship(r); + for (const n of bbNodes) sink.addNode(n); // BasicBlock layer → sink (CSV) + for (const e of pdgEdges) sink.addRelationship(e); + sink.finalize(); + const streamedCsvDir = path.join(tmpRoot, 'streamed'); + await streamAllCSVsToDisk(realGraph, path.join(tmpRoot, 'no-repo'), streamedCsvDir); + + // ── retention (R1): the real graph holds ZERO BasicBlocks ──────────── + let residentBasicBlocks = 0; + for (const n of realGraph.iterNodes()) if (n.label === 'BasicBlock') residentBasicBlocks++; + + // ── byte-identity (R2): per-file data-row set equality ─────────────── + const wholeBb = await dataRows(path.join(wholeDir, 'basicblock.csv')); + const streamedBb = await dataRows(path.join(tmpRoot, 'pdg-csv', 'basicblock.csv')); + const wholeRel = await dataRows(path.join(wholeDir, 'rel_BasicBlock_BasicBlock.csv')); + const streamedRel = await dataRows( + path.join(tmpRoot, 'pdg-csv', 'rel_BasicBlock_BasicBlock.csv'), + ); + + const bbIdentical = sha(wholeBb) === sha(streamedBb); + const relIdentical = sha(wholeRel) === sha(streamedRel); + // Fingerprint over the canonical PDG data-row set (drift gate). + const fingerprint = sha([...wholeBb, ...wholeRel].sort()); + + return { + scenario: 'streamingPdgEmit', + basic_blocks: bbNodes.length, + pdg_edges: pdgEdges.length, + chunk_rows: CHUNK_ROWS, + resident_basic_blocks: residentBasicBlocks, + byte_identical_nodes: bbIdentical, + byte_identical_edges: relIdentical, + fingerprint, + }; + } finally { + await fsp.rm(tmpRoot, { recursive: true, force: true }).catch(() => {}); + } +} + +const CHECK = process.argv.includes('--check'); +const result = await measure(); + +if (!CHECK) { + process.stdout.write(JSON.stringify(result) + '\n'); +} else { + const base = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8')); + const failures = []; + if (!result.byte_identical_nodes) + failures.push('streamed BasicBlock rows differ from whole-graph emit'); + if (!result.byte_identical_edges) + failures.push('streamed PDG-edge rows differ from whole-graph emit'); + if (result.resident_basic_blocks !== 0) { + failures.push( + `RSS bound violated: ${result.resident_basic_blocks} BasicBlock node(s) retained in the in-memory graph (expected 0)`, + ); + } + if (result.fingerprint !== base.fingerprint) { + failures.push(`fingerprint drift (got ${result.fingerprint}, expected ${base.fingerprint})`); + } + process.stdout.write(JSON.stringify(result) + '\n'); + if (failures.length > 0) { + for (const f of failures) process.stderr.write(`[stream-pdg-emit --check] FAIL: ${f}\n`); + process.exit(1); + } + process.stderr.write('[stream-pdg-emit --check] PASS\n'); +} diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index 334c022ac..ff01ccdb6 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -121,6 +121,23 @@ export interface PipelineOptions { /** Per-run `TAINT_PATH` edge cap (#2084 review P1-3). `undefined` ⇒ * `DEFAULT_PDG_MAX_INTERPROC_EDGES` (1000); `0` ⇒ no cap. */ pdgMaxInterprocEdges?: number; + /** + * Streaming/chunked PDG graph emit (#2202). When true, the BasicBlock + + * intra-file PDG-edge layer (CFG / REACHING_DEF / CDG / POST_DOMINATE / + * TAINTED / SANITIZES) is streamed to CSV-on-disk during the scope-resolution + * emit loop instead of being materialized in the in-memory graph, bounding + * peak RSS to O(chunk) rather than O(graph) at full-kernel scale. Already + * gated by the caller to full-rebuild runs only (the incremental writeback + * reads BasicBlocks back from the in-memory graph). Memory-only — produces a + * byte-identical persisted graph and is NOT part of `RepoMeta.pdg`, so + * toggling it never trips `pdgModeMismatch`. Default/false ⇒ today's + * whole-graph emit. + */ + streamPdgEmit?: boolean; + /** Streamed PDG-emit write buffer (rows) when `streamPdgEmit` is on (#2202). + * `undefined` ⇒ `DEFAULT_PDG_EMIT_CHUNK_ROWS`. Memory-only; does not affect + * emitted bytes. */ + pdgEmitChunkSize?: number; /** * Request parsing with the worker pool disabled. The sequential parser was * removed — the worker pool is the sole parse path — so setting this now @@ -287,10 +304,10 @@ export const runPipelineFromRepo = async ( let communityResult: CommunitiesOutput['communityResult'] | undefined; let processResult: ProcessesOutput['processResult'] | undefined; - const resolutionOutcomes = getPhaseOutput( - results, - 'scopeResolution', - ).resolutionOutcomes; + const scopeResolutionOutput = getPhaseOutput(results, 'scopeResolution'); + const resolutionOutcomes = scopeResolutionOutput.resolutionOutcomes; + // Streamed PDG-emit manifest (#2202): present only when streaming was on. + const pdgEmitManifest = scopeResolutionOutput.pdgEmitManifest; if (!options?.skipGraphPhases) { communityResult = getPhaseOutput(results, 'communities').communityResult; @@ -319,5 +336,6 @@ export const runPipelineFromRepo = async ( processResult, resolutionOutcomes, usedWorkerPool, + pdgEmitManifest, }; }; diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index 6bf86dc84..c7671d8f4 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -44,6 +44,8 @@ import { import type { ResolutionOutcome } from '../resolution-outcome.js'; import type { FunctionSummary } from '../../taint/summary-model.js'; import { buildFunctionNodeIndex } from '../../taint/summary-harvest-driver.js'; +import { PdgEmitSink, type PdgEmitManifest } from '../../../lbug/pdg-emit-sink.js'; +import { resolveNativeSafeStorageDir } from '../../../lbug/lbug-config.js'; import { logger } from '../../../logger.js'; export interface ScopeResolutionOutput { @@ -72,6 +74,14 @@ export interface ScopeResolutionOutput { * The `taintSummaries` phase composes these over the `CALLS` graph. */ readonly functionSummaries: readonly FunctionSummary[]; + /** + * Streamed PDG-emit COPY manifest (#2202). Present only when streaming was on + * (full rebuild + `--pdg` + enabled): the BasicBlock node CSV + per-pair PDG + * edge CSVs that were flushed to disk during the emit loop, for the persistence + * step to COPY alongside the structural CSVs. Absent ⇒ the PDG layer (if any) + * is in the in-memory graph and persists via the normal whole-graph emit. + */ + readonly pdgEmitManifest?: PdgEmitManifest; } const NOOP_OUTPUT: ScopeResolutionOutput = Object.freeze({ @@ -242,236 +252,291 @@ export const scopeResolutionPhase: PipelinePhase = { ? buildFunctionNodeIndex(ctx.graph) : undefined; - for (const [lang, provider] of SCOPE_RESOLVERS) { - // Standalone providers (COBOL, JCL) don't emit graph edges yet - // through the scope-resolution path. This is the canonical guard: - // runScopeResolution is never called for standalone providers, which - // keeps cobolPhase as the sole IMPORTS edge producer. Keep this guard - // in sync with any additional standalone providers added to - // SCOPE_RESOLVERS. - if (provider.languageProvider.parseStrategy === 'standalone') continue; - - const primaryLangFiles = filesByLang.get(lang) ?? []; - if (primaryLangFiles.length === 0) continue; - const primaryFilePaths = primaryLangFiles.map((f) => f.path); - - // Load per-language import-resolution config (tsconfig paths, - // composer.json autoload, go.mod, ...). One I/O round trip per - // workspace pass — cached implicitly by the result handed to - // every `resolveImportTarget` call below. - const resolutionConfig = - provider.loadResolutionConfig !== undefined - ? await provider.loadResolutionConfig(ctx.repoPath) - : undefined; - - // Some languages (e.g. Vue) expand their file universe beyond the - // primary-language files via the `collectScopeContextPaths` hook. - // The hook receives raw source contents of the primary files so it - // can trace import closures without a second tree-sitter parse. - // - // To avoid reading primary files twice (once for the hook, once for - // the resolution pass), we read them upfront and merge with the - // extra context paths the hook may add. - // Stream this language's pre-built ParsedFiles in from the disk store - // FIRST (huge-repo path). Doing it before reading source lets us skip - // loading content for files the store already covers — for a provider - // with no content-consuming hook that source is pure dead weight once - // extraction is served from the store (~1.5 GB on the kernel's C pass). - // Merged into `preExtractedByPath`; the per-language release block below - // evicts these again before the next language, so only one language's - // ParsedFiles are resident at a time. - const loadStoreFor = async (paths: ReadonlySet): Promise => { - if (!parsedFileStorePath) return; - const fromDisk = await loadParsedFilesForPaths(parsedFileStorePath, paths); - for (const [fp, pf] of fromDisk) preExtractedByPath.set(fp, pf); - }; - - // A provider that feeds source text into a post-extract hook - // (populateWorkspaceOwners / populateNamespaceSiblings / - // populateRangeBindings / emitPostResolutionEdges) needs content for ALL - // its files; one without those hooks only needs content for files the - // store does NOT cover (fresh-extract fallback). Keep this in sync with - // the getFileContents() call-sites in run.ts. - const providerNeedsAllContent = - provider.populateWorkspaceOwners !== undefined || - provider.populateNamespaceSiblings !== undefined || - provider.populateRangeBindings !== undefined || - provider.emitPostResolutionEdges !== undefined; - - let scopeFilePaths: Set; - let contents: Map; - if (provider.collectScopeContextPaths !== undefined) { - // Context-expanding providers (e.g. Vue) need every primary file's - // source up front for the closure hook, so load it all. - const entryFileContents = await readFileContents(ctx.repoPath, primaryFilePaths); - scopeFilePaths = provider.collectScopeContextPaths({ - primaryFilePaths, - preExtractedByPath, - entryFileContents, - allScannedPaths, - resolutionConfig, - }); - // Read only the extra context files (TS/JS etc.) not already loaded. - const extraPaths = [...scopeFilePaths].filter((p) => !entryFileContents.has(p)); - const extraContents = await readFileContents(ctx.repoPath, extraPaths); - contents = new Map([...entryFileContents, ...extraContents]); - await loadStoreFor(scopeFilePaths); + // Streaming/chunked PDG emit (#2202): when enabled (the caller has already + // gated this to full-rebuild + `--pdg`), route the BasicBlock + intra-file + // PDG-edge layer to CSV-on-disk through one sink shared across every + // language pass, so it never accumulates in `ctx.graph` (peak RSS O(chunk)). + // Needs the storage dir (the parse-cache store path, the same `.gitnexus` + // dir loadGraphToLbug COPYs from); if that is somehow absent we skip + // streaming and fall back to the in-memory whole-graph emit. + let pdgEmitSink: PdgEmitSink | undefined; + if (ctx.options?.streamPdgEmit === true && totalScopeFiles > 0) { + if (parsedFileStorePath) { + pdgEmitSink = new PdgEmitSink( + ctx.graph, + // Same ASCII-safe relocation the structural CSVs get (#2202 review #2): + // on Windows non-ASCII storage paths the COPY can't open files under + // the native path, so the dir is relocated to a hashed os.tmpdir(). + resolveNativeSafeStorageDir(parsedFileStorePath, 'pdg-csv'), + ctx.options?.pdgEmitChunkSize, + ); } else { - scopeFilePaths = new Set(primaryFilePaths); - await loadStoreFor(scopeFilePaths); - const pathsToRead = providerNeedsAllContent - ? primaryFilePaths - : primaryFilePaths.filter((p) => !preExtractedByPath.has(p)); - contents = await readFileContents(ctx.repoPath, pathsToRead); - } - const filePaths = [...scopeFilePaths]; - const files: { path: string; content: string }[] = []; - for (const fp of filePaths) { - const content = contents.get(fp); - if (content !== undefined) { - files.push({ path: fp, content }); - } else if (preExtractedByPath.has(fp)) { - // Store covers extraction for this file and we deliberately skipped - // reading its source; the empty string is never consumed (the - // extract loop uses the pre-extracted ParsedFile and this provider - // has no content hook). - files.push({ path: fp, content: '' }); - } - // else: uncovered AND unreadable → skip (unchanged from prior behavior). - } - - const langFileCount = files.length; - logHeapProbe( - 'scope-lang-start', - `lang=${lang} files=${langFileCount} contentsLoaded=${contents.size}`, - ); - const langLabel = lang.charAt(0).toUpperCase() + lang.slice(1); - currentLangIdx++; - const langTag = - totalScopeLangs > 1 ? `${langLabel} [${currentLangIdx}/${totalScopeLangs}]` : langLabel; - - if (totalScopeFiles > 0) { - const pct = - SCOPE_PCT_START + Math.round((processedScopeFiles / totalScopeFiles) * SCOPE_PCT_RANGE); - ctx.onProgress({ - phase: 'scopeResolution', - percent: pct, - message: 'Resolving types', - detail: `${langTag}, ${langFileCount.toLocaleString()} files`, - }); - } - - const stats = runScopeResolution( - { - graph: ctx.graph, - model, - files, - resolutionConfig, - prebuiltNodeLookup: sharedNodeLookup, - prebuiltFunctionNodeIndex: sharedFnNodeIndex, - preExtractedParsedFiles: preExtractedByPath, - scopeIndexStorePath: parsedFileStorePath, - // CFG/PDG emission (#2081 M1) — opt-in; off ⇒ byte-identical graph. - pdg: ctx.options?.pdg === true, - pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction, - pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction, - pdgMaxCdgEdgesPerFunction: ctx.options?.pdgMaxCdgEdgesPerFunction, - pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction, - pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops, - recordResolutionOutcome: (outcome) => { - resolutionOutcomes.push(outcome); - }, - onWarn: (msg) => { - if (isSemanticModelValidatorEnabled()) { - logger.warn(`[scope-resolution:${lang}] ${msg}`); - } - }, - onProgress: - totalScopeFiles > 0 - ? (subPhase: ScopeResolutionSubPhase, current, total) => { - let langRatio: number; - switch (subPhase) { - case 'extracting': - langRatio = total > 0 ? (current / total) * 0.5 : 0; - break; - case 'analyzing types': - langRatio = 0.5; - break; - case 'resolving references': - langRatio = 0.7; - break; - case 'linking symbols': - langRatio = 0.85; - break; - default: { - const _exhaustive: never = subPhase; - langRatio = 0.85; - } - } - const overallRatio = Math.min( - 1, - (processedScopeFiles + langRatio * langFileCount) / totalScopeFiles, - ); - const pct = SCOPE_PCT_START + Math.round(overallRatio * SCOPE_PCT_RANGE); - ctx.onProgress({ - phase: 'scopeResolution', - percent: pct, - message: 'Resolving types', - detail: - subPhase === 'extracting' - ? `${langTag} — extracting ${current.toLocaleString()}/${total.toLocaleString()} files` - : `${langTag} — ${subPhase}`, - }); - } - : undefined, - }, - provider, - ); - - // Release file contents and pre-extracted entries after each language - // to reduce memory pressure. For large codebases (16K+ PHP files), - // holding all source code simultaneously with scope trees causes OOM. - // See: https://github.com/abhigyanpatwari/GitNexus/issues/1741 - // - // Use `filePaths` (not `primaryFilePaths`) so that any context files - // added by `collectScopeContextPaths` (e.g. TS/JS files pulled in for - // Vue cross-file resolution) are also evicted and not held until GC. - files.length = 0; - contents.clear(); - for (const fp of filePaths) { - preExtractedByPath.delete(fp); - } - // This language's ParsedFiles are now unreachable (runScopeResolution has - // returned and the Map entries are deleted). Force a GC HERE so a heavy - // language's ~17-20GB set (e.g. C/C++ on the Linux kernel) is reclaimed - // BEFORE the next language's store-load — instead of leaving V8 to collect - // it lazily under the next pass's allocation pressure (which, at a cap >= - // RAM, degrades into swap-thrash). Collects only dead objects: the live - // cross-file index of the next pass is untouched. The pre/post probe - // confirms whether old-space fragmentation defeats the reclaim. - logHeapProbe('lang-release-pre-gc', `lang=${lang}`); - forceGc(); - logHeapProbe('lang-release-post-gc', `lang=${lang}`); - logHeapProbe('scope-lang-end', `lang=${lang} filesProcessed=${stats.filesProcessed}`); - - processedScopeFiles += langFileCount; - anyRan = true; - functionSummaries.push(...stats.functionSummaries); - totalFiles += stats.filesProcessed; - totalImports += stats.importsEmitted; - totalRefs += stats.referenceEdgesEmitted; - perLanguage.set(lang, { - filesProcessed: stats.filesProcessed, - importsEmitted: stats.importsEmitted, - referenceEdgesEmitted: stats.referenceEdgesEmitted, - }); - - if (isDev) { - logger.info( - `[scope-resolution:${lang}] ${stats.filesProcessed} files → ${stats.importsEmitted} IMPORTS + ${stats.referenceEdgesEmitted} reference edges (${stats.resolve.unresolved} unresolved sites, ${stats.referenceSkipped} skipped)`, + logger.warn( + '[scope-resolution] streaming PDG emit requested but no storage path is ' + + 'available; falling back to in-memory whole-graph emit', ); } } + // Cross-pass per-file dedup set for the streaming sink (#2202): one set + // shared across every language pass so a file emitted in two passes (e.g. a + // `.ts` module pulled into the Vue context pass) streams its PDG layer once. + // Only created when streaming — the in-memory-graph path dedups via its Map. + const pdgEmittedFiles = pdgEmitSink !== undefined ? new Set() : undefined; + + // Stream the PDG layer with guaranteed writer cleanup: a throw escaping the + // per-language loop (outside run.ts's per-file try/catch — e.g. from + // finalize/propagate/a provider hook) must still release the sink's file + // descriptors. finalize() runs on the success path; the finally closes the + // sink only when finalize did not (idempotent via the sink's `finalized`). + let pdgEmitManifest: PdgEmitManifest | undefined; + let pdgSinkSettled = false; + try { + for (const [lang, provider] of SCOPE_RESOLVERS) { + // Standalone providers (COBOL, JCL) don't emit graph edges yet + // through the scope-resolution path. This is the canonical guard: + // runScopeResolution is never called for standalone providers, which + // keeps cobolPhase as the sole IMPORTS edge producer. Keep this guard + // in sync with any additional standalone providers added to + // SCOPE_RESOLVERS. + if (provider.languageProvider.parseStrategy === 'standalone') continue; + + const primaryLangFiles = filesByLang.get(lang) ?? []; + if (primaryLangFiles.length === 0) continue; + const primaryFilePaths = primaryLangFiles.map((f) => f.path); + + // Load per-language import-resolution config (tsconfig paths, + // composer.json autoload, go.mod, ...). One I/O round trip per + // workspace pass — cached implicitly by the result handed to + // every `resolveImportTarget` call below. + const resolutionConfig = + provider.loadResolutionConfig !== undefined + ? await provider.loadResolutionConfig(ctx.repoPath) + : undefined; + + // Some languages (e.g. Vue) expand their file universe beyond the + // primary-language files via the `collectScopeContextPaths` hook. + // The hook receives raw source contents of the primary files so it + // can trace import closures without a second tree-sitter parse. + // + // To avoid reading primary files twice (once for the hook, once for + // the resolution pass), we read them upfront and merge with the + // extra context paths the hook may add. + // Stream this language's pre-built ParsedFiles in from the disk store + // FIRST (huge-repo path). Doing it before reading source lets us skip + // loading content for files the store already covers — for a provider + // with no content-consuming hook that source is pure dead weight once + // extraction is served from the store (~1.5 GB on the kernel's C pass). + // Merged into `preExtractedByPath`; the per-language release block below + // evicts these again before the next language, so only one language's + // ParsedFiles are resident at a time. + const loadStoreFor = async (paths: ReadonlySet): Promise => { + if (!parsedFileStorePath) return; + const fromDisk = await loadParsedFilesForPaths(parsedFileStorePath, paths); + for (const [fp, pf] of fromDisk) preExtractedByPath.set(fp, pf); + }; + + // A provider that feeds source text into a post-extract hook + // (populateWorkspaceOwners / populateNamespaceSiblings / + // populateRangeBindings / emitPostResolutionEdges) needs content for ALL + // its files; one without those hooks only needs content for files the + // store does NOT cover (fresh-extract fallback). Keep this in sync with + // the getFileContents() call-sites in run.ts. + const providerNeedsAllContent = + provider.populateWorkspaceOwners !== undefined || + provider.populateNamespaceSiblings !== undefined || + provider.populateRangeBindings !== undefined || + provider.emitPostResolutionEdges !== undefined; + + let scopeFilePaths: Set; + let contents: Map; + if (provider.collectScopeContextPaths !== undefined) { + // Context-expanding providers (e.g. Vue) need every primary file's + // source up front for the closure hook, so load it all. + const entryFileContents = await readFileContents(ctx.repoPath, primaryFilePaths); + scopeFilePaths = provider.collectScopeContextPaths({ + primaryFilePaths, + preExtractedByPath, + entryFileContents, + allScannedPaths, + resolutionConfig, + }); + // Read only the extra context files (TS/JS etc.) not already loaded. + const extraPaths = [...scopeFilePaths].filter((p) => !entryFileContents.has(p)); + const extraContents = await readFileContents(ctx.repoPath, extraPaths); + contents = new Map([...entryFileContents, ...extraContents]); + await loadStoreFor(scopeFilePaths); + } else { + scopeFilePaths = new Set(primaryFilePaths); + await loadStoreFor(scopeFilePaths); + const pathsToRead = providerNeedsAllContent + ? primaryFilePaths + : primaryFilePaths.filter((p) => !preExtractedByPath.has(p)); + contents = await readFileContents(ctx.repoPath, pathsToRead); + } + const filePaths = [...scopeFilePaths]; + const files: { path: string; content: string }[] = []; + for (const fp of filePaths) { + const content = contents.get(fp); + if (content !== undefined) { + files.push({ path: fp, content }); + } else if (preExtractedByPath.has(fp)) { + // Store covers extraction for this file and we deliberately skipped + // reading its source; the empty string is never consumed (the + // extract loop uses the pre-extracted ParsedFile and this provider + // has no content hook). + files.push({ path: fp, content: '' }); + } + // else: uncovered AND unreadable → skip (unchanged from prior behavior). + } + + const langFileCount = files.length; + logHeapProbe( + 'scope-lang-start', + `lang=${lang} files=${langFileCount} contentsLoaded=${contents.size}`, + ); + const langLabel = lang.charAt(0).toUpperCase() + lang.slice(1); + currentLangIdx++; + const langTag = + totalScopeLangs > 1 ? `${langLabel} [${currentLangIdx}/${totalScopeLangs}]` : langLabel; + + if (totalScopeFiles > 0) { + const pct = + SCOPE_PCT_START + Math.round((processedScopeFiles / totalScopeFiles) * SCOPE_PCT_RANGE); + ctx.onProgress({ + phase: 'scopeResolution', + percent: pct, + message: 'Resolving types', + detail: `${langTag}, ${langFileCount.toLocaleString()} files`, + }); + } + + const stats = runScopeResolution( + { + graph: ctx.graph, + model, + files, + resolutionConfig, + prebuiltNodeLookup: sharedNodeLookup, + prebuiltFunctionNodeIndex: sharedFnNodeIndex, + preExtractedParsedFiles: preExtractedByPath, + scopeIndexStorePath: parsedFileStorePath, + // CFG/PDG emission (#2081 M1) — opt-in; off ⇒ byte-identical graph. + pdg: ctx.options?.pdg === true, + pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction, + pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction, + pdgMaxCdgEdgesPerFunction: ctx.options?.pdgMaxCdgEdgesPerFunction, + pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction, + pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops, + // Streaming PDG-emit sink (#2202) — undefined ⇒ emit to the in-memory graph. + pdgEmitSink, + // Cross-pass per-file dedup set (#2202) — undefined when not streaming. + pdgEmittedFiles, + recordResolutionOutcome: (outcome) => { + resolutionOutcomes.push(outcome); + }, + onWarn: (msg) => { + if (isSemanticModelValidatorEnabled()) { + logger.warn(`[scope-resolution:${lang}] ${msg}`); + } + }, + onProgress: + totalScopeFiles > 0 + ? (subPhase: ScopeResolutionSubPhase, current, total) => { + let langRatio: number; + switch (subPhase) { + case 'extracting': + langRatio = total > 0 ? (current / total) * 0.5 : 0; + break; + case 'analyzing types': + langRatio = 0.5; + break; + case 'resolving references': + langRatio = 0.7; + break; + case 'linking symbols': + langRatio = 0.85; + break; + default: { + const _exhaustive: never = subPhase; + langRatio = 0.85; + } + } + const overallRatio = Math.min( + 1, + (processedScopeFiles + langRatio * langFileCount) / totalScopeFiles, + ); + const pct = SCOPE_PCT_START + Math.round(overallRatio * SCOPE_PCT_RANGE); + ctx.onProgress({ + phase: 'scopeResolution', + percent: pct, + message: 'Resolving types', + detail: + subPhase === 'extracting' + ? `${langTag} — extracting ${current.toLocaleString()}/${total.toLocaleString()} files` + : `${langTag} — ${subPhase}`, + }); + } + : undefined, + }, + provider, + ); + + // Release file contents and pre-extracted entries after each language + // to reduce memory pressure. For large codebases (16K+ PHP files), + // holding all source code simultaneously with scope trees causes OOM. + // See: https://github.com/abhigyanpatwari/GitNexus/issues/1741 + // + // Use `filePaths` (not `primaryFilePaths`) so that any context files + // added by `collectScopeContextPaths` (e.g. TS/JS files pulled in for + // Vue cross-file resolution) are also evicted and not held until GC. + files.length = 0; + contents.clear(); + for (const fp of filePaths) { + preExtractedByPath.delete(fp); + } + // This language's ParsedFiles are now unreachable (runScopeResolution has + // returned and the Map entries are deleted). Force a GC HERE so a heavy + // language's ~17-20GB set (e.g. C/C++ on the Linux kernel) is reclaimed + // BEFORE the next language's store-load — instead of leaving V8 to collect + // it lazily under the next pass's allocation pressure (which, at a cap >= + // RAM, degrades into swap-thrash). Collects only dead objects: the live + // cross-file index of the next pass is untouched. The pre/post probe + // confirms whether old-space fragmentation defeats the reclaim. + logHeapProbe('lang-release-pre-gc', `lang=${lang}`); + forceGc(); + logHeapProbe('lang-release-post-gc', `lang=${lang}`); + logHeapProbe('scope-lang-end', `lang=${lang} filesProcessed=${stats.filesProcessed}`); + + processedScopeFiles += langFileCount; + anyRan = true; + functionSummaries.push(...stats.functionSummaries); + totalFiles += stats.filesProcessed; + totalImports += stats.importsEmitted; + totalRefs += stats.referenceEdgesEmitted; + perLanguage.set(lang, { + filesProcessed: stats.filesProcessed, + importsEmitted: stats.importsEmitted, + referenceEdgesEmitted: stats.referenceEdgesEmitted, + }); + + if (isDev) { + logger.info( + `[scope-resolution:${lang}] ${stats.filesProcessed} files → ${stats.importsEmitted} IMPORTS + ${stats.referenceEdgesEmitted} reference edges (${stats.resolve.unresolved} unresolved sites, ${stats.referenceSkipped} skipped)`, + ); + } + } + + // Finalize the streaming PDG sink (#2202) once after the last language: + // flush + close its CSV writers and capture the COPY manifest. forceGc at + // the boundary reclaims transient write buffers (mirrors the per-language + // release below). + pdgEmitManifest = pdgEmitSink?.finalize(); + pdgSinkSettled = true; + if (pdgEmitSink !== undefined) forceGc(); + } finally { + // Release fds if a throw skipped finalize (idempotent with finalize()). + if (pdgEmitSink !== undefined && !pdgSinkSettled) pdgEmitSink.close(); + } if (totalScopeFiles > 0 && anyRan) { ctx.onProgress({ @@ -494,7 +559,10 @@ export const scopeResolutionPhase: PipelinePhase = { } } - if (!anyRan) return NOOP_OUTPUT; + // Even when no language ran, surface a finalized manifest (its CSVs are on + // disk) so loadGraphToLbug COPYs them rather than orphaning them — empty in + // the no-files case, harmless. + if (!anyRan) return pdgEmitManifest ? { ...NOOP_OUTPUT, pdgEmitManifest } : NOOP_OUTPUT; return { ran: true, @@ -504,6 +572,7 @@ export const scopeResolutionPhase: PipelinePhase = { resolutionOutcomes, perLanguage, functionSummaries, + pdgEmitManifest, }; }, }; diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts index aa30c2416..7b573e0a6 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts @@ -302,6 +302,30 @@ interface RunScopeResolutionInput { * `reason`; consumed by the U4 taint emit step). `undefined` ⇒ * `DEFAULT_PDG_MAX_TAINT_HOPS` (32); `0` ⇒ no cap. */ readonly pdgMaxTaintHops?: number; + /** + * Streaming PDG-emit sink (#2202). When present (streaming on, full rebuild), + * the `--pdg` emit routes BasicBlock nodes + intra-file PDG edges to THIS + * graph-shaped target instead of the in-memory `graph`, so the bulky PDG + * layer never accumulates in memory (peak RSS O(chunk)). Typed as a plain + * `KnowledgeGraph` so this module stays decoupled from the persistence layer; + * the caller (the scope-resolution phase) owns its lifecycle and finalizes it + * after the last language. Absent ⇒ the emit writes to `graph` as before + * (byte-identical default). + */ + readonly pdgEmitSink?: KnowledgeGraph; + /** + * Cross-pass per-file dedup set for streaming PDG emit (#2202). Shared across + * every language pass (owned by the scope-resolution phase). A file imported + * by more than one language (e.g. a `.ts` module pulled into the Vue context + * pass) is PDG-emitted in each pass over the same `cfgSideChannel`, producing + * identical ids; the in-memory graph dedups that by id, but the streaming sink + * is dedup-free (to stay O(write buffer), not O(total ids)). So when present + * (streaming on), the emit loop skips a file whose PDG already streamed and + * records the rest — keeping the streamed set byte-identical to the + * Map-deduped whole-graph emit, for any language-pass order. Absent ⇒ no skip + * (the graph Map dedups), so the default path is unchanged. + */ + readonly pdgEmittedFiles?: Set; /** * Optional graph-node lookup built ONCE by the caller and shared across * every language pass. `buildGraphNodeLookup` scans the whole graph and is @@ -769,6 +793,11 @@ export function runScopeResolution( // can bracket it. Printed as the PROF `taint=` segment. let taintMs = 0; if (input.pdg === true) { + // Streaming target (#2202): when a sink is provided, BasicBlock nodes + + // intra-file PDG edges are routed to CSV-on-disk through it instead of + // accumulating in `graph`. The function-node index below is still built + // from the real `graph` (Function/Method nodes live there, never the sink). + const pdgTarget: KnowledgeGraph = input.pdgEmitSink ?? graph; let cfgBlocks = 0; let cfgEdges = 0; let cfgDroppedEdges = 0; @@ -835,6 +864,15 @@ export function runScopeResolution( // shard that slipped the version gate) must skip emission, not throw a // TypeError mid-graph-build and abort scope-resolution for the language. if (!Array.isArray(cfgs) || cfgs.length === 0) continue; + // Cross-pass per-file dedup (#2202): when streaming, a file whose PDG + // already streamed in a prior language pass (e.g. a `.ts` module pulled + // into the Vue context pass) would re-emit identical ids from the same + // cfgSideChannel — the dedup-free streaming sink would double the rows. + // Skip it here; the in-memory-graph path needs no skip (its Map dedups). + if (input.pdgEmittedFiles !== undefined) { + if (input.pdgEmittedFiles.has(pf.filePath)) continue; + input.pdgEmittedFiles.add(pf.filePath); + } try { // Per-element emit-safety filter (mirrors the parsedfile-store // reviver's POLICY: valid elements in a mixed array still emit; junk @@ -854,7 +892,7 @@ export function runScopeResolution( } if (wellFormed.length === 0) continue; const emitted = emitFileCfgs( - graph, + pdgTarget, wellFormed, input.pdgMaxEdgesPerFunction ?? DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, // Log cap-overflow drops UNCONDITIONALLY (not via input.onWarn, which is @@ -873,7 +911,7 @@ export function runScopeResolution( // PROF-gated like every other checkpoint here (zero cost when off). const t0 = PROF ? performance.now() : 0; const rd = emitFileReachingDefs( - graph, + pdgTarget, wellFormed, input.pdgMaxReachingDefEdgesPerFunction ?? DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, @@ -892,7 +930,7 @@ export function runScopeResolution( // persisted and its time folds into the `pdg=` PROF segment next to RD. const tCdg = PROF ? performance.now() : 0; const cdg = emitFileCdg( - graph, + pdgTarget, wellFormed, input.pdgMaxCdgEdgesPerFunction ?? DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, (message) => logger.warn(message), // unconditional — R6, no silent truncation @@ -909,7 +947,7 @@ export function runScopeResolution( if (taintSpec !== undefined) { const t1 = PROF ? performance.now() : 0; const taint = emitFileTaint( - graph, + pdgTarget, wellFormed, pf.parsedImports, taintSpec, diff --git a/gitnexus/src/core/ingestion/utils/env.ts b/gitnexus/src/core/ingestion/utils/env.ts index 5beeb818f..70b3baa3f 100644 --- a/gitnexus/src/core/ingestion/utils/env.ts +++ b/gitnexus/src/core/ingestion/utils/env.ts @@ -28,6 +28,18 @@ export const parseTruthyEnv = (raw: string | undefined): boolean => { return value === '1' || value === 'true' || value === 'yes'; }; +/** + * Parse a positive-integer env-var value. Returns the integer when `raw` is a + * finite integer `> 0`; otherwise (`undefined`, empty, non-numeric, `0`, or + * negative) returns `undefined` so the caller falls back to its default. Used + * for numeric tuning knobs like `GITNEXUS_PDG_EMIT_CHUNK_SIZE` (#2202). + */ +export const parsePositiveIntEnv = (raw: string | undefined): number | undefined => { + if (raw === undefined) return undefined; + const n = Number(raw.trim()); + return Number.isInteger(n) && n > 0 ? n : undefined; +}; + /** * Whether scope-resolution dev validators (e.g. `validateBindingsImmutability`) * should run AND emit warnings. Off by default in CLI runs to avoid silent diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index 955a2d47c..af756c74f 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -248,6 +248,24 @@ export const buildRelRow = (rel: GraphRelationship): string => escapeCSVNumber(rel.step, 0), ].join(','); +/** Canonical BasicBlock node CSV header — taint/PDG substrate (issue #2080). + * No `name` column; blocks are identified by id + source span. Shared by the + * whole-graph emit pass and the streaming PDG emit sink (issue #2202) so the + * two paths produce byte-identical BasicBlock rows by construction. */ +export const BASICBLOCK_CSV_HEADER = 'id,filePath,startLine,endLine,text'; + +/** Build the escaped CSV row (no trailing newline) for one BasicBlock node. + * Single source of the BasicBlock row bytes — used by `streamAllCSVsToDisk` + * and by the streaming `PdgEmitSink` (issue #2202). */ +export const buildBasicBlockRow = (node: GraphNode): string => + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.filePath || ''), + escapeCSVNumber(node.properties.startLine, -1), + escapeCSVNumber(node.properties.endLine, -1), + escapeCSVField(node.properties.text || ''), + ].join(','); + export interface StreamedCSVResult { nodeFiles: Map; /** pairKey (`From|To`) → per-FROM→TO-label-pair CSV file. */ @@ -345,7 +363,7 @@ export const streamAllCSVsToDisk = async ( // blocks are identified by id + source span. Emitted by no phase yet. const basicBlockWriter = new BufferedCSVWriter( path.join(csvDir, 'basicblock.csv'), - 'id,filePath,startLine,endLine,text', + BASICBLOCK_CSV_HEADER, ); // Multi-language node types share the same CSV shape (no isExported column) @@ -526,15 +544,7 @@ export const streamAllCSVsToDisk = async ( ); break; case 'BasicBlock': - pending = basicBlockWriter.addRow( - [ - escapeCSVField(node.id), - escapeCSVField(node.properties.filePath || ''), - escapeCSVNumber(node.properties.startLine, -1), - escapeCSVNumber(node.properties.endLine, -1), - escapeCSVField(node.properties.text || ''), - ].join(','), - ); + pending = basicBlockWriter.addRow(buildBasicBlockRow(node)); break; default: { // Code element nodes (Function, Class, Interface, CodeElement) diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index 457c49175..1b0c08738 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -4,8 +4,6 @@ import { createInterface } from 'readline'; import { once } from 'events'; import { finished } from 'stream/promises'; import path from 'path'; -import os from 'os'; -import crypto from 'crypto'; import lbug from '@ladybugdb/core'; import { closeQueryResults } from './query-result-utils.js'; import { KnowledgeGraph } from '../graph/types.js'; @@ -19,6 +17,7 @@ import { NodeTableName, } from './schema.js'; import { streamAllCSVsToDisk } from './csv-generator.js'; +import type { PdgEmitManifest } from './pdg-emit-sink.js'; import { getNodeLabel as deriveNodeLabel, type WriteStreamFactory } from './rel-pair-routing.js'; import type { CachedEmbedding } from '../embeddings/types.js'; import { extensionManager, type ExtensionEnsureOptions } from './extension-loader.js'; @@ -29,6 +28,7 @@ import { isWalCorruptionError, openLbugConnection, toNativeSafePath, + resolveNativeSafeStorageDir, WAL_RECOVERY_SUGGESTION, waitForWindowsHandleRelease, type LbugConnectionHandle, @@ -881,6 +881,15 @@ export const loadGraphToLbug = async ( repoPath: string, storagePath: string, onProgress?: LbugProgressCallback, + /** + * Streamed PDG-emit manifest (#2202). When present (streaming was on, full + * rebuild), the BasicBlock node CSV + per-pair PDG-edge CSVs it points at + * were already flushed to disk during the emit loop; they are merged into the + * COPY plan below so they load alongside the structural CSVs. When streaming + * was on the in-memory `graph` holds zero BasicBlocks, so `streamAllCSVsToDisk` + * emits none — the manifest is the sole source and there is no double-COPY. + */ + pdgEmitManifest?: PdgEmitManifest, ) => { if (!conn) { throw new Error('LadybugDB not initialized. Call initLbug first.'); @@ -899,16 +908,43 @@ export const loadGraphToLbug = async ( const span = (a: bigint, b: bigint): string => (Number(b - a) / 1e6).toFixed(1); const tStart = mark(); - let csvDir: string; - if (process.platform === 'win32' && /[^\x00-\x7F]/.test(storagePath)) { - const hash = crypto.createHash('sha256').update(storagePath).digest('hex').slice(0, 16); - csvDir = toNativeSafePath(path.join(os.tmpdir(), `gitnexus-csv-${hash}`)); - } else { - csvDir = path.join(storagePath, 'csv'); - } + const csvDir = resolveNativeSafeStorageDir(storagePath, 'csv'); log('Streaming CSVs to disk...'); const csvResult = await streamAllCSVsToDisk(graph, repoPath, csvDir); + + // Merge the streamed PDG-emit CSVs (#2202) into the COPY plan so the + // BasicBlock node table + per-pair PDG edges (CFG / REACHING_DEF / CDG / + // POST_DOMINATE / TAINTED / SANITIZES) load through the SAME node + per-pair + // COPY loops as the structural CSVs. The graph held zero BasicBlocks when + // streaming, so `streamAllCSVsToDisk` produced none of these — the manifest + // is the sole source and there is no double-COPY. Absent ⇒ no-op. + if (pdgEmitManifest) { + for (const [table, meta] of pdgEmitManifest.nodeFiles) { + // A collision means a BasicBlock leaked into the in-memory graph during a + // streamed run (streamAllCSVsToDisk then emitted a structural basicblock.csv). + // That is a streaming-invariant violation — fail loudly rather than + // silently overwrite one CSV with the other and drop its rows (#2202 review #3). + if (csvResult.nodeFiles.has(table)) { + throw new Error( + `Streaming PDG manifest collides with a structural node CSV for "${table}" — ` + + `the in-memory graph should hold zero ${table} nodes when streaming. ` + + `A ${table} node leaked into the graph during a streamed emit.`, + ); + } + csvResult.nodeFiles.set(table, meta); + } + for (const [pairKey, meta] of pdgEmitManifest.relsByPair) { + if (csvResult.relsByPair.has(pairKey)) { + throw new Error( + `Streaming PDG manifest collides with a structural relationship CSV for pair ` + + `"${pairKey}" — a PDG edge leaked into the in-memory graph during a streamed emit.`, + ); + } + csvResult.relsByPair.set(pairKey, meta); + csvResult.totalValidRels += meta.rows; + } + } const tCsv = mark(); const validTables = new Set(NODE_TABLES as readonly string[]); diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index aeba61d6a..46c4605be 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -189,6 +189,36 @@ export function toNativeSafePath(p: string): string { return p; } +/** + * Resolve the on-disk CSV staging dir for `/`, applying the + * same ASCII-safe relocation `toNativeSafePath` enables: on Windows with a + * non-ASCII storage path, LadybugDB's bulk COPY cannot open files under that + * path, so the dir is relocated under `os.tmpdir()`. Shared by the structural + * `csv/` dir and the streaming `pdg-csv/` dir (#2202) so the two can never + * diverge on platform handling; the `gitnexus--` prefix keeps their tmp + * locations distinct and recognizable. + * + * The relocated dir is created with `fs.mkdtempSync` (a unique, mode-0700, + * guaranteed-not-pre-existing suffix) rather than a deterministic + * `gitnexus--` name. A predictable name in the world-readable OS + * temp dir is information-disclosure-prone and pre-plantable + * (CWE-377/378 / CodeQL `js/insecure-temporary-file`); mkdtemp's random suffix + * is the documented mitigation and is what reaches the streaming sink's + * `fs.openSync`. The non-Windows / ASCII path stays a pure `path.join` (no dir + * created) and is byte-identical to before. + */ +export function resolveNativeSafeStorageDir(storagePath: string, subdir: string): string { + if (process.platform === 'win32' && NON_ASCII_RE.test(storagePath)) { + // 8.3-shorten the tmpdir base first (a non-ASCII Windows *profile* path can + // make os.tmpdir() itself non-ASCII), THEN mkdtemp so the returned path — + // the one that flows into fs.openSync — is provably mkdtemp-sourced (random, + // exclusive) and clears the insecure-temp-file dataflow. + const base = toNativeSafePath(os.tmpdir()); + return fsSync.mkdtempSync(path.join(base, `gitnexus-${subdir}-`)); + } + return path.join(storagePath, subdir); +} + /** * Shared configuration for `@ladybugdb/core` `Database` construction. * diff --git a/gitnexus/src/core/lbug/pdg-emit-sink.ts b/gitnexus/src/core/lbug/pdg-emit-sink.ts new file mode 100644 index 000000000..79cc05f58 --- /dev/null +++ b/gitnexus/src/core/lbug/pdg-emit-sink.ts @@ -0,0 +1,395 @@ +/** + * Streaming PDG graph-emit sink (issue #2202). + * + * The PDG emit loop (`scope-resolution/pipeline/run.ts`, the `--pdg` block) + * materializes BasicBlock nodes + intra-file PDG edges (CFG / REACHING_DEF / + * CDG / POST_DOMINATE / TAINTED / SANITIZES) into the in-memory + * `KnowledgeGraph`. At full-kernel scale that layer dominates peak RSS + * (~7 GB at 511K BasicBlocks; ~100 GB extrapolated to the full kernel → OOM). + * + * `PdgEmitSink` is a write-routing façade over the real graph: the emit + * functions are write-only and compute every edge endpoint by deterministic id + * (audited — no read-back), so the sink can route BasicBlock node rows and PDG + * edge rows straight to bounded CSV-on-disk writers and **never store them**. + * The graph's resident size stops growing with the PDG layer → peak RSS becomes + * O(chunk buffer), not O(graph). Everything else (structural nodes/edges, the + * whole-program M4 TAINT_PATH edges) is delegated to the real graph unchanged. + * + * Why synchronous writers? The whole PDG emit (`runScopeResolution` and its + * per-file loop) is synchronous — there is no `await` point to drain an async + * stream, so a `BufferedCSVWriter` (Node `WriteStream`) would accumulate + * unwritten chunks in process memory across millions of rows, defeating the RSS + * bound. `fs.writeSync` goes straight to the OS; resident memory is bounded to + * one `chunkRows` buffer. This mirrors the sync-shard pattern in + * `storage/parsedfile-store.ts`. + * + * Byte-identity (issue acceptance): the sink reuses the SAME shared row + * builders (`buildBasicBlockRow`, `buildRelRow`) and label derivation + * (`getNodeLabel`) as `streamAllCSVsToDisk`, so the streamed CSV line SET is + * identical to the whole-graph emit's, and the bulk COPY loads the same rows → + * the persisted graph is SET-identical and DB-identical. The guarantee is + * set-level, not byte-level on the CSV file: the sink streams rows in emit + * order and does NOT re-sort them under `GITNEXUS_SORT_GRAPH_OUTPUT`, so a + * streamed CSV file is not necessarily byte-for-byte equal to the sorted + * whole-graph CSV — but the row set, and therefore the DB outcome, is. (The + * streamed CSVs are deleted right after the COPY, so their on-disk byte order + * is never observed.) Cross-pass dedup is done upstream, per FILE, in the emit + * loop (`run.ts` skips a file whose PDG already streamed) rather than in the + * sink, because a file can be PDG-emitted in more than one language pass (a + * `.ts` module imported by a `.vue` SFC is emitted in both the TypeScript and + * Vue context passes over the same worker-built CFG) and a sink-level per-id + * dedup set would retain every id → O(total ids) memory, defeating the + * O(chunk) RSS bound (#2202 review #1). The differential fingerprint test + * (issue #2202 U6) and the Vue+TS cross-pass integration test guard the set. + */ + +import fs from 'fs'; +import path from 'path'; +import type { GraphNode, GraphRelationship, RelationshipType } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../graph/types.js'; +import { + BASICBLOCK_CSV_HEADER, + REL_CSV_HEADER, + buildBasicBlockRow, + buildRelRow, +} from './csv-generator.js'; +import { getNodeLabel } from './rel-pair-routing.js'; +import { NODE_TABLES, type NodeTableName } from './schema.js'; + +/** + * PDG edge types streamed per-file (all intra-block BasicBlock→BasicBlock). + * `TAINT_PATH` is intentionally excluded — it is the whole-program M4 edge + * (Function→Function), computed in a separate post-resolution phase over the + * complete CALLS graph, and stays in the in-memory graph (it is small and is + * persisted by the normal whole-graph emit). + */ +const PDG_EDGE_TYPES: ReadonlySet = new Set([ + 'CFG', + 'REACHING_DEF', + 'CDG', + 'POST_DOMINATE', + 'TAINTED', + 'SANITIZES', +]); + +/** Default streamed-write buffer (rows). Matches the whole-graph emit's + * `FLUSH_EVERY` order of magnitude; overridable via `GITNEXUS_PDG_EMIT_CHUNK_SIZE`. */ +export const DEFAULT_PDG_EMIT_CHUNK_ROWS = 500; + +/** + * Synchronous buffered CSV writer. Buffers up to `chunkRows` rows, then issues + * one `fs.writeSync` straight to the OS (no in-process stream buffer). Header + * is written into the buffer at construction and is NOT counted in `rows` + * (matching `BufferedCSVWriter` semantics, so manifest row counts line up). + */ +class SyncCsvWriter { + private fd: number; + private buf: string[] = []; + private readonly chunkRows: number; + rows = 0; + /** + * First IO error this writer hit (a `fs.writeSync` short-write loop throwing + * on e.g. disk-full). Once poisoned the writer refuses further rows and + * skips its final flush; the sink surfaces it from {@link PdgEmitSink.finalize} + * so a truncated CSV is never handed to the bulk COPY (#2202 review #4). A + * streamed-write failure is an IO fault, not the CFG-logic error that the + * emit loop's per-file try/catch is built to swallow — poisoning routes it + * past that catch to a loud failure. + */ + poison: unknown | undefined = undefined; + + constructor( + readonly csvPath: string, + header: string, + chunkRows: number, + ) { + // Guard a 0/negative buffer: the flush modulo would never fire and `buf` + // would grow unbounded, defeating the whole point of streaming. + this.chunkRows = Math.max(1, chunkRows); + // Exclusive create (O_EXCL): the streamed-CSV dir is wiped + recreated fresh + // by the PdgEmitSink constructor before any writer opens a file, so the path + // never pre-exists — 'wx' both matches that invariant and refuses to follow + // a pre-planted symlink at the path (CWE-377 / CodeQL js/insecure-temporary-file). + this.fd = fs.openSync(csvPath, 'wx'); + this.buf.push(header); + } + + addRow(row: string): void { + // A poisoned writer is dead — stop buffering so memory can't grow on a + // writer whose fd is already in a bad state; finalize will report the fault. + if (this.poison !== undefined) return; + this.buf.push(row); + this.rows++; + // Flush on DATA-row count, not buffer length: the header occupies buf[0] + // until the first flush, so a `buf.length >= chunkRows` test would fire one + // row early on the first chunk. Counting rows makes every flush exactly + // `chunkRows` rows. + if (this.rows % this.chunkRows === 0) this.flushOrPoison(); + } + + /** Flush, recording (and re-throwing) any IO error as poison. Re-throwing + * lets the immediate caller log the per-file failure; the persisted `poison` + * is the backstop that makes finalize fail loudly even when that throw is + * swallowed by the emit loop's CFG try/catch. */ + private flushOrPoison(): void { + try { + this.flush(); + } catch (e) { + this.poison ??= e; + throw e; + } + } + + private flush(): void { + if (this.buf.length === 0) return; + const data = Buffer.from(this.buf.join('\n') + '\n', 'utf8'); + // fs.writeSync can return a short byte count; loop until the whole buffer + // lands so a partial write never truncates a CSV row mid-field. + let offset = 0; + while (offset < data.length) { + offset += fs.writeSync(this.fd, data, offset, data.length - offset); + } + this.buf.length = 0; + } + + /** Flush remaining rows (unless already poisoned) and close the fd. Never + * throws: a final-flush IO error is recorded as poison and the fd is still + * closed, so a write error neither leaks an fd nor escapes here — the sink + * reads {@link poison} after closing every writer and fails loudly then. */ + close(): void { + try { + if (this.poison === undefined) this.flush(); + } catch (e) { + this.poison ??= e; + } finally { + try { + fs.closeSync(this.fd); + } catch { + /* fd may already be invalid after an IO fault — nothing to recover */ + } + } + } +} + +/** + * COPY manifest produced by {@link PdgEmitSink.finalize}. Shaped to merge + * directly into `StreamedCSVResult` so `loadGraphToLbug` COPYs the streamed + * PDG CSVs through the same per-table / per-pair loops as the structural CSVs. + * Paths are absolute, so persistence needs no dir recomputation. + */ +export interface PdgEmitManifest { + /** Node-table CSVs (only `BasicBlock` today). */ + readonly nodeFiles: Map; + /** pairKey (`From|To`) → per-pair edge CSV. */ + readonly relsByPair: Map; +} + +/** + * Write-routing graph façade. Construct one per analyze run, thread it into the + * per-language `runScopeResolution` calls in place of the real graph during the + * `--pdg` emit, then {@link finalize} once after the last language. + */ +export class PdgEmitSink implements KnowledgeGraph { + private readonly validTables: Set; + private bbWriter: SyncCsvWriter | undefined; + /** pairKey (`From|To`) → writer. PDG edges are all `BasicBlock|BasicBlock`, + * but the map keeps the sink general and the manifest pair-keyed. */ + private readonly relWriters = new Map(); + private finalized = false; + /** + * First writer-construction failure (a `fs.openSync` throwing on e.g. EMFILE + * — out of file descriptors). The failure happens inside the `SyncCsvWriter` + * constructor before a writer object exists to carry poison, so it is held + * here at the sink level and folded into the {@link finalize} error check. + * Like an in-flight write fault, an open failure mid-emit would otherwise be + * swallowed by the emit loop's per-file try/catch and silently drop the rest + * of that file's rows (#2202 review #4/#6). + */ + private openFailure: unknown | undefined = undefined; + // NOTE on dedup: the same file can be PDG-emitted in more than one language + // pass (e.g. a `.ts` module imported by a `.vue` SFC is emitted in both the + // TypeScript pass and the Vue context pass over the same worker-built + // `cfgSideChannel`). The in-memory graph dedups that by id (first-writer-wins); + // this sink does NOT — to keep peak memory O(write buffer) rather than + // O(total ids), cross-pass dedup is done upstream, per FILE, in the emit loop + // (`run.ts` skips a file whose PDG already streamed via `pdgEmittedFiles`). + // The sink therefore receives each id exactly once and is a faithful + // pass-through; it must not be fed duplicate ids. + + constructor( + private readonly real: KnowledgeGraph, + private readonly pdgCsvDir: string, + private readonly chunkRows: number = DEFAULT_PDG_EMIT_CHUNK_ROWS, + ) { + this.validTables = new Set(NODE_TABLES as readonly string[]); + // Clear any streamed CSVs left by a previous (possibly crashed) run so a + // later COPY never picks up stale rows. + fs.rmSync(pdgCsvDir, { recursive: true, force: true }); + fs.mkdirSync(pdgCsvDir, { recursive: true }); + } + + // ── routed writes ────────────────────────────────────────────────────────── + + addNode(node: GraphNode): void { + if (node.label === 'BasicBlock') { + if (this.bbWriter === undefined) { + try { + this.bbWriter = new SyncCsvWriter( + path.join(this.pdgCsvDir, 'basicblock.csv'), + BASICBLOCK_CSV_HEADER, + this.chunkRows, + ); + } catch (e) { + this.openFailure ??= e; + throw e; + } + } + this.bbWriter.addRow(buildBasicBlockRow(node)); + return; + } + this.real.addNode(node); + } + + addRelationship(relationship: GraphRelationship): void { + if (PDG_EDGE_TYPES.has(relationship.type)) { + const fromLabel = getNodeLabel(relationship.sourceId); + const toLabel = getNodeLabel(relationship.targetId); + // Skip edges whose endpoint labels are not valid node tables — mirrors + // `RelPairRouter` exactly so the streamed set matches the whole-graph set. + if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) return; + const pairKey = `${fromLabel}|${toLabel}`; + let writer = this.relWriters.get(pairKey); + if (writer === undefined) { + try { + writer = new SyncCsvWriter( + path.join(this.pdgCsvDir, `rel_${fromLabel}_${toLabel}.csv`), + REL_CSV_HEADER, + this.chunkRows, + ); + } catch (e) { + this.openFailure ??= e; + throw e; + } + this.relWriters.set(pairKey, writer); + } + writer.addRow(buildRelRow(relationship)); + return; + } + this.real.addRelationship(relationship); + } + + /** Flush + close every streamed writer and return the COPY manifest. Every + * fd is closed even when a writer is poisoned (its `close` never throws); any + * IO fault — an in-flight write that poisoned a writer, a final-flush failure, + * or a writer-open failure (EMFILE) — is surfaced loudly here so a disk-full + * / out-of-fds run never hands a truncated CSV to the bulk COPY (#2202 review + * #4). The emit loop's per-file try/catch swallows the synchronous throw, so + * this poison check is the backstop that turns a silent partial manifest into + * a hard failure. */ + finalize(): PdgEmitManifest { + if (this.finalized) throw new Error('PdgEmitSink.finalize() called twice'); + this.finalized = true; + + const errors: unknown[] = []; + if (this.openFailure !== undefined) errors.push(this.openFailure); + + const nodeFiles = new Map(); + if (this.bbWriter !== undefined) { + this.bbWriter.close(); + if (this.bbWriter.poison !== undefined) errors.push(this.bbWriter.poison); + nodeFiles.set('BasicBlock' as NodeTableName, { + csvPath: this.bbWriter.csvPath, + rows: this.bbWriter.rows, + }); + } + + const relsByPair = new Map(); + for (const [pairKey, writer] of this.relWriters) { + writer.close(); + if (writer.poison !== undefined) errors.push(writer.poison); + relsByPair.set(pairKey, { csvPath: writer.csvPath, rows: writer.rows }); + } + + if (errors.length > 0) { + const first = errors[0]; + throw new Error( + `PdgEmitSink: ${errors.length} streamed CSV writer(s) hit an IO error ` + + `(disk-full / out-of-fds) during the emit — the persisted graph would ` + + `be truncated, so the run is failed rather than COPYing a partial CSV: ${ + first instanceof Error ? first.message : String(first) + }`, + ); + } + + return { nodeFiles, relsByPair }; + } + + /** + * Best-effort fd release for the error path — when a language pass throws + * before {@link finalize} runs, the caller's `finally` calls this so the + * BasicBlock + per-pair fds never leak. Idempotent with finalize via the + * `finalized` flag; close errors are swallowed because the run is already + * failing. + */ + close(): void { + if (this.finalized) return; + this.finalized = true; + try { + this.bbWriter?.close(); + } catch { + /* best-effort */ + } + for (const writer of this.relWriters.values()) { + try { + writer.close(); + } catch { + /* best-effort */ + } + } + } + + // ── delegated reads / non-PDG mutations ───────────────────────────────────── + // The PDG emit functions never call these on the routed graph, but the + // façade implements the full KnowledgeGraph surface so it is a drop-in for + // the emit target and any non-PDG write transparently reaches the real graph. + + get nodes(): GraphNode[] { + return this.real.nodes; + } + get relationships(): GraphRelationship[] { + return this.real.relationships; + } + iterNodes(): IterableIterator { + return this.real.iterNodes(); + } + iterRelationships(): IterableIterator { + return this.real.iterRelationships(); + } + iterRelationshipsByType(type: RelationshipType): IterableIterator { + return this.real.iterRelationshipsByType(type); + } + forEachNode(fn: (node: GraphNode) => void): void { + this.real.forEachNode(fn); + } + forEachRelationship(fn: (rel: GraphRelationship) => void): void { + this.real.forEachRelationship(fn); + } + getNode(id: string): GraphNode | undefined { + return this.real.getNode(id); + } + get nodeCount(): number { + return this.real.nodeCount; + } + get relationshipCount(): number { + return this.real.relationshipCount; + } + removeNode(nodeId: string): boolean { + return this.real.removeNode(nodeId); + } + removeNodesByFile(filePath: string): number { + return this.real.removeNodesByFile(filePath); + } + removeRelationship(relationshipId: string): boolean { + return this.real.removeRelationship(relationshipId); + } +} diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 769125ad5..ed92c9b9c 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -61,6 +61,7 @@ import { } from './ingestion/taint/interproc-solver.js'; import { DEFAULT_PDG_MAX_INTERPROC_EDGES } from './ingestion/taint/interproc-emit.js'; import { taintModelVersion } from './ingestion/taint/typescript-model.js'; +import { parseTruthyEnv, parsePositiveIntEnv } from './ingestion/utils/env.js'; import { computeFileHashes, diffFileHashes } from '../storage/file-hash.js'; import { extractChangedSubgraph, @@ -170,6 +171,19 @@ export interface AnalyzeOptions { pdgMaxInterprocFindings?: number; pdgMaxInterprocHops?: number; pdgMaxInterprocEdges?: number; + /** + * Stream the BasicBlock + intra-file PDG-edge layer to CSV-on-disk during the + * emit loop instead of materializing it in the in-memory graph, bounding peak + * RSS to O(chunk) for full-kernel-scale repos (#2202). Only engages on a full + * rebuild — `resolveStreamPdgEmit` additionally requires `force === true` + * (the pre-pipeline guarantee of a full rebuild). May also be enabled via + * `GITNEXUS_STREAM_PDG_EMIT`. Memory-only; byte-identical output; not stamped + * into `RepoMeta.pdg`. */ + streamPdgEmit?: boolean; + /** Streamed PDG-emit write buffer (rows). `undefined` ⇒ + * `DEFAULT_PDG_EMIT_CHUNK_ROWS`. May also be set via + * `GITNEXUS_PDG_EMIT_CHUNK_SIZE`. Memory-only (#2202). */ + pdgEmitChunkSize?: number; /** * Default branch threaded into generated AGENTS.md / CLAUDE.md so the * regression-compare example uses the configured branch instead of a @@ -426,6 +440,48 @@ export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] => } : undefined; +/** + * Whether streaming/chunked PDG graph emit (#2202) engages this run. + * + * Streaming flushes the BasicBlock + intra-file PDG-edge layer to CSV-on-disk + * during the emit loop and never lands it in the in-memory graph, bounding peak + * RSS to O(chunk). It is sound ONLY on a full rebuild: the incremental + * writeback (`extractChangedSubgraph`) reads BasicBlock nodes back out of the + * in-memory graph, which streaming has already offloaded. `force === true` is + * the pre-pipeline guarantee of a full rebuild — `isIncremental` has + * `!force` as a necessary condition — so gating on it avoids the deliberately + * absent pre-pipeline incremental prediction (see the `isIncremental` note). + * + * Requires `pdg === true` (nothing to stream otherwise). Enabled by either the + * explicit `streamPdgEmit` option or the `GITNEXUS_STREAM_PDG_EMIT` env toggle. + * Memory-only — NOT part of {@link resolvePdgConfig}, so toggling it never + * trips `pdgModeMismatch`. Read every call (not memoized) so `vi.stubEnv` + * works in tests. Pure + exported for testing. + */ +export const resolveStreamPdgEmit = (options: { + pdg?: boolean; + force?: boolean; + streamPdgEmit?: boolean; +}): boolean => + options.pdg === true && + options.force === true && + (options.streamPdgEmit === true || parseTruthyEnv(process.env.GITNEXUS_STREAM_PDG_EMIT)); + +/** + * Resolve the streamed PDG-emit write-buffer size (#2202). Explicit option wins + * over `GITNEXUS_PDG_EMIT_CHUNK_SIZE`; `undefined` ⇒ the sink's + * `DEFAULT_PDG_EMIT_CHUNK_ROWS`. Memory-only; does not affect emitted bytes. + */ +export const resolvePdgEmitChunkSize = (options: { + pdgEmitChunkSize?: number; +}): number | undefined => { + // Only honor a positive-integer explicit option; `0`/negative is NOT nullish + // so `?? env` would pass it through and make the sink flush every row. + const opt = options.pdgEmitChunkSize; + if (opt !== undefined && Number.isInteger(opt) && opt > 0) return opt; + return parsePositiveIntEnv(process.env.GITNEXUS_PDG_EMIT_CHUNK_SIZE); +}; + /** * Whether the requested `--pdg` configuration differs from the one the * existing index's DB rows were built under (#2099 F1). An absent recorded @@ -828,6 +884,11 @@ export async function runFullAnalysis( pdgMaxInterprocFindings: options.pdgMaxInterprocFindings, pdgMaxInterprocHops: options.pdgMaxInterprocHops, pdgMaxInterprocEdges: options.pdgMaxInterprocEdges, + // Streaming/chunked PDG emit (#2202) — gated to full-rebuild runs + // (force === true) so the incremental writeback never reads back an + // offloaded BasicBlock layer. Memory-only; byte-identical output. + streamPdgEmit: resolveStreamPdgEmit(options), + pdgEmitChunkSize: resolvePdgEmitChunkSize(options), fetchWrappers: options.fetchWrappers, }, ); @@ -1069,11 +1130,21 @@ export async function runFullAnalysis( }); } else { // ── Full rebuild ─────────────────────────────────────────────── - await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => { - lbugMsgCount++; - const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24)); - progress('lbug', pct, msg); - }); + // Pass the streamed PDG-emit manifest (#2202) so the BasicBlock layer that + // was flushed to CSV during the emit loop is COPY'd alongside the + // structural CSVs. Only ever set on a full rebuild (streaming is + // force-gated), so the incremental branch above never carries it. + await loadGraphToLbug( + pipelineResult.graph, + pipelineResult.repoPath, + storagePath, + (msg) => { + lbugMsgCount++; + const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24)); + progress('lbug', pct, msg); + }, + pipelineResult.pdgEmitManifest, + ); } // ── Phase 3: FTS (85–90%) ───────────────────────────────────────── diff --git a/gitnexus/src/types/pipeline.ts b/gitnexus/src/types/pipeline.ts index 5ad08cae6..4cbb28886 100644 --- a/gitnexus/src/types/pipeline.ts +++ b/gitnexus/src/types/pipeline.ts @@ -2,6 +2,7 @@ import type { KnowledgeGraph } from '../core/graph/types.js'; import { CommunityDetectionResult } from '../core/ingestion/community-processor.js'; import { ProcessDetectionResult } from '../core/ingestion/process-processor.js'; import type { ResolutionOutcome } from '../core/ingestion/scope-resolution/resolution-outcome.js'; +import type { PdgEmitManifest } from '../core/lbug/pdg-emit-sink.js'; // CLI-specific: in-memory result with graph + detection results export interface PipelineResult { @@ -27,4 +28,12 @@ export interface PipelineResult { * affordance so regression suites can prove the pool engaged. */ usedWorkerPool: boolean; + /** + * Streamed PDG-emit COPY manifest (#2202). Present only when streaming/chunked + * PDG emit was active (full rebuild + `--pdg` + enabled): the BasicBlock node + * CSV + per-pair PDG-edge CSVs flushed to disk during the emit loop, for + * `loadGraphToLbug` to COPY alongside the structural CSVs. Absent ⇒ the PDG + * layer (if any) is resident in `graph` and persists via the whole-graph emit. + */ + pdgEmitManifest?: PdgEmitManifest; } diff --git a/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue b/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue new file mode 100644 index 000000000..b7c100e4a --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue @@ -0,0 +1,28 @@ + + + + diff --git a/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts b/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts new file mode 100644 index 000000000..6bb7197c9 --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts @@ -0,0 +1,44 @@ +// Shared TypeScript module imported by app.vue. Because app.vue's +// `collectScopeContextPaths` does a transitive import closure, THIS file is +// pulled into the Vue scope-resolution pass IN ADDITION to the primary +// TypeScript pass — so its worker-built CFG (the functions below) is +// PDG-emitted in BOTH passes over the same `cfgSideChannel`, producing +// identical BasicBlock + PDG-edge ids. The in-memory graph dedups those by id +// (first-writer-wins Map); the streaming sink relies on per-file dedup in +// run.ts (`pdgEmittedFiles`). This is the real cross-pass double-emit the +// #2202 streaming dedup must collapse (review #8a). + +export function classify(x: number): string { + let label: string; + if (x > 0) { + label = 'positive'; + } else if (x < 0) { + label = 'negative'; + } else { + label = 'zero'; + } + return label; +} + +export function accumulate(n: number): number { + let sum = 0; + for (let i = 0; i < n; i++) { + if (i % 2 === 0) { + sum += i; + } else { + sum -= 1; + } + } + return sum; +} + +export function guard(value: number): number { + if (value > 100) { + return clamp(value); + } + return value; +} + +function clamp(value: number): number { + return value > 100 ? 100 : value; +} diff --git a/gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts b/gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts new file mode 100644 index 000000000..ddce07071 --- /dev/null +++ b/gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts @@ -0,0 +1,167 @@ +/** + * End-to-end proof of streaming/chunked PDG graph emit (issue #2202). + * + * Runs the real pipeline (workers + scope-resolution) on the pdg-repo fixture + * TWICE — a non-streamed baseline and a streamed run — and asserts: + * - the streamed run's in-memory graph holds ZERO BasicBlock nodes and ZERO + * intra-file PDG edges (the bulky layer was flushed to CSV, never resident + * — the O(chunk) RSS bound, R1); + * - the streamed run produces a `pdgEmitManifest` whose BasicBlock + PDG-edge + * row counts EQUAL the baseline's resident counts (same emitted SET, R2); + * - the baseline (streaming off) still emits the PDG layer into the graph, + * i.e. the default path is unchanged (R3). + * + * Both runs use a durable parse cache (streaming requires `parseCache.storagePath` + * for its CSV dir), differing ONLY in `streamPdgEmit`, so streaming is the only + * variable. + */ +import { describe, it, expect, afterAll } from 'vitest'; +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { runPipelineFromRepo } from '../../../src/core/ingestion/pipeline.js'; +import { loadParseCache } from '../../../src/storage/parse-cache.js'; +import type { PipelineResult } from '../../../src/types/pipeline.js'; + +const FIXTURE = path.join(__dirname, 'fixtures', 'pdg-repo'); +// A `.vue` SFC importing a `.ts` module: the TS module is PDG-emitted in BOTH +// the TypeScript pass and the Vue context pass (review #8a, cross-pass dedup). +const VUE_TS_FIXTURE = path.join(__dirname, 'fixtures', 'vue-ts-pdg'); +const PDG_EDGE_TYPES = new Set([ + 'CFG', + 'REACHING_DEF', + 'CDG', + 'POST_DOMINATE', + 'TAINTED', + 'SANITIZES', +]); + +const tmpDirs: string[] = []; +function freshRepo(fixture: string = FIXTURE): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-pdg-stream-')); + fs.cpSync(fixture, dir, { recursive: true }); + tmpDirs.push(dir); + return dir; +} +function freshStorage(): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-pdg-store-')); + tmpDirs.push(dir); + return dir; +} + +function pdgCounts(result: PipelineResult): { basicBlocks: number; pdgEdges: number } { + let basicBlocks = 0; + result.graph.forEachNode((n) => { + if (n.label === 'BasicBlock') basicBlocks++; + }); + let pdgEdges = 0; + for (const rel of result.graph.iterRelationships()) { + if (PDG_EDGE_TYPES.has(rel.type)) pdgEdges++; + } + return { basicBlocks, pdgEdges }; +} + +describe('#2202 — streaming PDG emit end-to-end', () => { + afterAll(() => { + for (const d of tmpDirs) fs.rmSync(d, { recursive: true, force: true }); + }); + + it('streams the PDG layer out of the graph while preserving the emitted set', async () => { + // ── Baseline: --pdg on, streaming OFF (durable cache, same as streamed) ── + const baseStorage = freshStorage(); + const baseline = await runPipelineFromRepo(freshRepo(), () => {}, { + pdg: true, + parseCache: await loadParseCache(baseStorage), + }); + const base = pdgCounts(baseline); + // R3 / sanity: the default path still materializes the PDG layer in-graph. + expect(base.basicBlocks).toBeGreaterThan(0); + expect(base.pdgEdges).toBeGreaterThan(0); + expect(baseline.pdgEmitManifest).toBeUndefined(); + + // ── Streamed: --pdg on, streaming ON ───────────────────────────────── + const streamStorage = freshStorage(); + const streamed = await runPipelineFromRepo(freshRepo(), () => {}, { + pdg: true, + streamPdgEmit: true, + parseCache: await loadParseCache(streamStorage), + }); + const streamedCounts = pdgCounts(streamed); + + // R1: the bulky PDG layer never accumulated in the in-memory graph. + expect(streamedCounts.basicBlocks).toBe(0); + expect(streamedCounts.pdgEdges).toBe(0); + + // R2: the streamed manifest carries the SAME emitted set as the baseline. + const manifest = streamed.pdgEmitManifest; + expect(manifest).toBeDefined(); + const bbRows = manifest!.nodeFiles.get('BasicBlock')?.rows ?? 0; + expect(bbRows).toBe(base.basicBlocks); + let manifestEdgeRows = 0; + for (const [, meta] of manifest!.relsByPair) manifestEdgeRows += meta.rows; + expect(manifestEdgeRows).toBe(base.pdgEdges); + + // The streamed BasicBlock CSV exists on disk under the storage dir. + const bbCsv = manifest!.nodeFiles.get('BasicBlock')?.csvPath; + expect(bbCsv).toBeDefined(); + expect(fs.existsSync(bbCsv!)).toBe(true); + }); + + it('collapses the real Vue+TS cross-pass double-emit to one streamed copy (review #8a)', async () => { + // A `.ts` module imported by a `.vue` SFC is PDG-emitted in BOTH the + // TypeScript pass and the Vue context pass (the Vue provider's + // `collectScopeContextPaths` follows the import) over the same worker-built + // `cfgSideChannel` → identical ids. The in-memory graph dedups those by id + // (Map first-writer-wins); the streaming sink is dedup-free, so the emit + // loop dedups per FILE via `pdgEmittedFiles`. Without that dedup the streamed + // manifest would carry shared.ts's blocks TWICE (verified out-of-band: 33 → + // 61 BasicBlock rows). This asserts the streamed SET equals the Map-deduped + // baseline — the load-bearing dedup regression guard. + const baseStorage = freshStorage(); + const baseline = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, { + pdg: true, + parseCache: await loadParseCache(baseStorage), + }); + const base = pdgCounts(baseline); + // Both files contribute blocks — the cross-pass case is actually present. + expect(base.basicBlocks).toBeGreaterThan(0); + expect(base.pdgEdges).toBeGreaterThan(0); + + const streamStorage = freshStorage(); + const streamed = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, { + pdg: true, + streamPdgEmit: true, + parseCache: await loadParseCache(streamStorage), + }); + // Streamed graph holds none of the PDG layer. + const streamedCounts = pdgCounts(streamed); + expect(streamedCounts.basicBlocks).toBe(0); + expect(streamedCounts.pdgEdges).toBe(0); + + // The streamed manifest carries each file's PDG layer EXACTLY ONCE — equal + // to the Map-deduped baseline. A broken per-file dedup would double the + // shared module's rows and fail here. + const manifest = streamed.pdgEmitManifest; + expect(manifest).toBeDefined(); + expect(manifest!.nodeFiles.get('BasicBlock')?.rows ?? 0).toBe(base.basicBlocks); + let manifestEdgeRows = 0; + for (const [, meta] of manifest!.relsByPair) manifestEdgeRows += meta.rows; + expect(manifestEdgeRows).toBe(base.pdgEdges); + }); + + it('falls back to in-memory emit when streaming is on but no storage path exists', async () => { + // `streamPdgEmit: true` with NO parse cache → `parsedFileStorePath` is + // undefined, so phase.ts cannot place the streamed CSV dir and falls back to + // the in-memory whole-graph emit (the `else` branch that warns). The PDG + // layer must still land in the graph and NO manifest is produced. + const fellBack = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, { + pdg: true, + streamPdgEmit: true, + // intentionally no parseCache → no storagePath + }); + const counts = pdgCounts(fellBack); + expect(counts.basicBlocks).toBeGreaterThan(0); // emitted in-memory, not streamed + expect(counts.pdgEdges).toBeGreaterThan(0); + expect(fellBack.pdgEmitManifest).toBeUndefined(); // no streaming happened + }); +}); diff --git a/gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts b/gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts new file mode 100644 index 000000000..d29d9f524 --- /dev/null +++ b/gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts @@ -0,0 +1,181 @@ +/** + * Integration test: streamed PDG-emit manifest round-trips the bulk-COPY load + * path (issue #2202 U5). + * + * Simulates the streaming case end-to-end at the persistence boundary: the + * BasicBlock + intra-file PDG-edge layer is flushed to CSV by a real + * `PdgEmitSink` (so the in-memory graph holds ZERO BasicBlocks, exactly as in a + * streamed run), and `loadGraphToLbug` is handed the resulting manifest. Asserts + * the BasicBlock nodes + every PDG edge type land in the DB via the manifest, + * alongside the structural graph — and that there is no double-COPY. + */ +import { describe, it, expect, beforeAll, afterAll } from 'vitest'; +import fs from 'fs/promises'; +import path from 'path'; +import os from 'os'; + +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import { PdgEmitSink } from '../../src/core/lbug/pdg-emit-sink.js'; +import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; + +let tmpBase: string; +let storagePath: string; + +const FILE_ID = 'File:src/a.ts'; +const BB = (i: number) => `BasicBlock:src/a.ts:1:0:${i}`; +const PDG_TYPES = ['CFG', 'REACHING_DEF', 'CDG', 'POST_DOMINATE', 'TAINTED', 'SANITIZES'] as const; + +beforeAll(async () => { + // mkdtemp (unpredictable, unique) — not a predictable os-temp path. + tmpBase = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-pdg-stream-rt-')); + storagePath = path.join(tmpBase, '.gitnexus'); + await fs.mkdir(path.join(storagePath, 'lbug'), { recursive: true }); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.initLbug(path.join(storagePath, 'lbug')); + + // Structural graph — NO BasicBlock nodes (they were "streamed out"). + const graph = createKnowledgeGraph(); + graph.addNode({ + id: FILE_ID, + label: 'File', + properties: { name: 'a.ts', filePath: 'src/a.ts' }, + }); + + // Real sink → real manifest: route 3 BasicBlocks + one edge of each PDG type. + const sink = new PdgEmitSink(graph, path.join(storagePath, 'pdg-csv')); + for (let i = 0; i < 3; i++) { + const node: GraphNode = { + id: BB(i), + label: 'BasicBlock', + properties: { + name: '', + filePath: 'src/a.ts', + startLine: i + 1, + endLine: i + 2, + text: `b${i}`, + }, + }; + sink.addNode(node); + } + for (const type of PDG_TYPES) { + const rel: GraphRelationship = { + id: `${type}:0->1`, + sourceId: BB(0), + targetId: BB(1), + type, + confidence: 1, + reason: type === 'REACHING_DEF' ? 'x' : type === 'CDG' ? 'T' : `${type}-edge`, + }; + sink.addRelationship(rel); + } + const manifest = sink.finalize(); + + // The sink offloaded the whole PDG layer — the graph has only the File node. + expect(graph.nodeCount).toBe(1); + + await adapter.loadGraphToLbug(graph, tmpBase, storagePath, undefined, manifest); +}); + +afterAll(async () => { + try { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.closeLbug(); + } catch { + /* may not have opened */ + } + if (tmpBase) { + for (let attempt = 0; attempt < 5; attempt++) { + try { + await fs.rm(tmpBase, { recursive: true, force: true }); + return; + } catch { + if (attempt < 4) await new Promise((r) => setTimeout(r, 200 * (attempt + 1))); + } + } + } +}); + +describe('streamed PDG manifest → bulk COPY (#2202 U5)', () => { + it('BasicBlock nodes from the manifest land in the DB with span + text', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const rows = await adapter.executeQuery( + 'MATCH (n:BasicBlock) RETURN n.id AS id, n.text AS text, n.startLine AS startLine ORDER BY n.id', + ); + expect(rows).toHaveLength(3); + expect(rows[0].id).toBe(BB(0)); + expect(rows[0].text).toBe('b0'); + expect(Number(rows[0].startLine)).toBe(1); + }); + + it('the structural graph (File node) loaded alongside the manifest', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const rows = await adapter.executeQuery( + `MATCH (f:File {id: '${FILE_ID}'}) RETURN count(f) AS c`, + ); + expect(Number(rows[0].c)).toBe(1); + }); + + it('every PDG edge type round-trips via the manifest (no double-COPY)', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + for (const type of PDG_TYPES) { + const rows = await adapter.executeQuery( + `MATCH (:BasicBlock)-[r:CodeRelation {type: '${type}'}]->(:BasicBlock) RETURN count(r) AS c`, + ); + // Exactly one — not two (double-COPY would double these). + expect(Number(rows[0].c), `${type} should round-trip exactly once`).toBe(1); + } + }); + + it('REACHING_DEF carries its variable in reason (manifest path)', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const rows = await adapter.executeQuery( + "MATCH (a:BasicBlock)-[r:CodeRelation {type: 'REACHING_DEF', reason: 'x'}]->(b:BasicBlock) RETURN a.id AS from, b.id AS to", + ); + expect(rows).toHaveLength(1); + expect(rows[0].from).toBe(BB(0)); + expect(rows[0].to).toBe(BB(1)); + }); +}); + +describe('streamed PDG manifest → disjoint-key merge guard (#2202 review #3)', () => { + // The merge in loadGraphToLbug assumes the streamed manifest and the + // structural csvResult are disjoint: when streaming is on the in-memory graph + // holds ZERO BasicBlocks, so streamAllCSVsToDisk emits no basicblock.csv and + // the manifest is the sole source. A future BasicBlock-leak-into-graph would + // make both sides carry a "BasicBlock" entry; silently overwriting one CSV + // with the other would drop its rows. The guard fails loudly instead. + + it('throws when the manifest collides with a structural node CSV', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + // A graph that DOES contain a BasicBlock → streamAllCSVsToDisk emits a + // structural basicblock.csv (the invariant-violation scenario). + const leakyGraph = createKnowledgeGraph(); + leakyGraph.addNode({ + id: BB(0), + label: 'BasicBlock', + properties: { name: '', filePath: 'src/a.ts', startLine: 1, endLine: 2, text: 'leak' }, + }); + + // A manifest that ALSO declares a BasicBlock node CSV → disjoint-key clash. + const collideBase = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-pdg-collide-')); + const collideStorage = path.join(collideBase, '.gitnexus'); + await fs.mkdir(collideStorage, { recursive: true }); + const sink = new PdgEmitSink(createKnowledgeGraph(), path.join(collideStorage, 'pdg-csv')); + sink.addNode({ + id: BB(1), + label: 'BasicBlock', + properties: { name: '', filePath: 'src/a.ts', startLine: 3, endLine: 4, text: 'm' }, + }); + const manifest = sink.finalize(); + + try { + await expect( + adapter.loadGraphToLbug(leakyGraph, collideBase, collideStorage, undefined, manifest), + ).rejects.toThrow(/collides with a structural node CSV for "BasicBlock"/); + } finally { + await fs.rm(collideBase, { recursive: true, force: true }); + } + }); +}); diff --git a/gitnexus/test/unit/lbug-native-safe-path.test.ts b/gitnexus/test/unit/lbug-native-safe-path.test.ts index 5d13b6a36..48d3c4711 100644 --- a/gitnexus/test/unit/lbug-native-safe-path.test.ts +++ b/gitnexus/test/unit/lbug-native-safe-path.test.ts @@ -5,7 +5,12 @@ * 8.3 short-name form before passing them to KuzuDB's native layer. */ import { describe, it, expect } from 'vitest'; -import { toNativeSafePath, cleanupNativePathJunctions } from '../../src/core/lbug/lbug-config.js'; +import path from 'path'; +import { + toNativeSafePath, + cleanupNativePathJunctions, + resolveNativeSafeStorageDir, +} from '../../src/core/lbug/lbug-config.js'; describe('toNativeSafePath', () => { it('returns ASCII paths unchanged on any platform', () => { @@ -61,3 +66,52 @@ describe('toNativeSafePath', () => { }); } }); + +describe('resolveNativeSafeStorageDir (#2202)', () => { + it('returns / for an ASCII storage path on any platform', () => { + const storage = path.join('repo', '.gitnexus'); + expect(resolveNativeSafeStorageDir(storage, 'csv')).toBe(path.join(storage, 'csv')); + expect(resolveNativeSafeStorageDir(storage, 'pdg-csv')).toBe(path.join(storage, 'pdg-csv')); + }); + + it('keeps csv and pdg-csv distinct (no collision between structural and streamed dirs)', () => { + const storage = path.join('repo', '.gitnexus'); + expect(resolveNativeSafeStorageDir(storage, 'csv')).not.toBe( + resolveNativeSafeStorageDir(storage, 'pdg-csv'), + ); + }); + + if (process.platform !== 'win32') { + it('does NOT relocate a non-ASCII storage path off Windows (platform gate)', () => { + const storage = path.join('repo', '用户', '.gitnexus'); + // Non-win32: the relocation never fires regardless of non-ASCII chars. + expect(resolveNativeSafeStorageDir(storage, 'pdg-csv')).toBe(path.join(storage, 'pdg-csv')); + }); + } + + if (process.platform === 'win32') { + it('relocates a non-ASCII storage path to a unique mkdtemp os.tmpdir() dir per subdir', () => { + const fs = require('fs'); + const storage = 'C:\\Project\\中文\\.gitnexus'; + // mkdtemp creates the dirs — track + clean them up. + const csv = resolveNativeSafeStorageDir(storage, 'csv'); + const pdg = resolveNativeSafeStorageDir(storage, 'pdg-csv'); + try { + // Both relocated under os.tmpdir(), ASCII-prefixed, and distinct (each + // mkdtemp call returns a fresh random suffix — never a predictable name). + expect(pdg.includes('gitnexus-pdg-csv-')).toBe(true); + expect(csv.includes('gitnexus-csv-')).toBe(true); + expect(csv).not.toBe(pdg); + // Two calls for the same (storage, subdir) yield DIFFERENT dirs (random). + const csv2 = resolveNativeSafeStorageDir(storage, 'csv'); + expect(csv2).not.toBe(csv); + // The relocated paths are not under the original non-ASCII storage path. + expect(pdg.includes('中文')).toBe(false); + fs.rmSync(csv2, { recursive: true, force: true }); + } finally { + fs.rmSync(csv, { recursive: true, force: true }); + fs.rmSync(pdg, { recursive: true, force: true }); + } + }); + } +}); diff --git a/gitnexus/test/unit/lbug/pdg-emit-sink.test.ts b/gitnexus/test/unit/lbug/pdg-emit-sink.test.ts new file mode 100644 index 000000000..7d4699529 --- /dev/null +++ b/gitnexus/test/unit/lbug/pdg-emit-sink.test.ts @@ -0,0 +1,331 @@ +/** + * PdgEmitSink unit tests (issue #2202 U2). + * + * Verifies the streaming PDG emit sink: + * - routes BasicBlock nodes + PDG edges to bounded CSV-on-disk; + * - delegates structural nodes/edges + the whole-program TAINT_PATH edge to + * the real graph (never streamed); + * - is byte-identical (set-wise) to the whole-graph `streamAllCSVsToDisk` + * emit for the same node/edge set (the issue's byte-identity acceptance); + * - never accumulates the PDG layer in the in-memory graph (the RSS bound). + */ +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'; +import fs from 'node:fs'; +import fsp from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; + +import { createKnowledgeGraph } from '../../../src/core/graph/graph.js'; +import { streamAllCSVsToDisk, buildBasicBlockRow } from '../../../src/core/lbug/csv-generator.js'; +import { PdgEmitSink } from '../../../src/core/lbug/pdg-emit-sink.js'; +import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; + +const bbNode = (fp: string, idx: number, line: number): GraphNode => ({ + id: `BasicBlock:${fp}:1:0:${idx}`, + label: 'BasicBlock', + properties: { name: '', filePath: fp, startLine: line, endLine: line + 1, text: `blk ${idx}` }, +}); + +const pdgEdge = ( + fp: string, + from: number, + to: number, + type: GraphRelationship['type'], + reason: string, +): GraphRelationship => ({ + id: `${type}:${fp}:${from}->${to}`, + sourceId: `BasicBlock:${fp}:1:0:${from}`, + targetId: `BasicBlock:${fp}:1:0:${to}`, + type, + confidence: 1, + reason, +}); + +/** Sorted non-empty lines of a CSV file (order-independent comparison). */ +const sortedLines = async (csvPath: string): Promise => { + const text = await fsp.readFile(csvPath, 'utf8'); + return text + .split('\n') + .filter((l) => l.length > 0) + .sort(); +}; + +let tmpRoot: string; + +beforeEach(() => { + tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'pdg-sink-')); +}); + +afterEach(() => { + fs.rmSync(tmpRoot, { recursive: true, force: true }); +}); + +describe('PdgEmitSink — routing', () => { + it('routes BasicBlock nodes and PDG edges to CSV, never to the real graph', () => { + const real = createKnowledgeGraph(); + const sink = new PdgEmitSink(real, path.join(tmpRoot, 'pdg-csv')); + + sink.addNode(bbNode('a.ts', 0, 1)); + sink.addNode(bbNode('a.ts', 1, 5)); + sink.addRelationship(pdgEdge('a.ts', 0, 1, 'CFG', 'seq')); + sink.addRelationship(pdgEdge('a.ts', 0, 1, 'REACHING_DEF', 'x:1:0')); + + // PDG layer must not land in the in-memory graph (the RSS bound). + expect(real.nodeCount).toBe(0); + expect(real.relationshipCount).toBe(0); + expect(sink.nodeCount).toBe(0); + + sink.finalize(); + }); + + it('delegates structural nodes, CALLS, and the whole-program TAINT_PATH edge to the real graph', () => { + const real = createKnowledgeGraph(); + const sink = new PdgEmitSink(real, path.join(tmpRoot, 'pdg-csv')); + + sink.addNode({ + id: 'Function:a.ts:fn:1', + label: 'Function', + properties: { name: 'fn', filePath: 'a.ts', startLine: 1, endLine: 9 }, + }); + sink.addRelationship({ + id: 'CALLS:1', + sourceId: 'Function:a.ts:fn:1', + targetId: 'Function:a.ts:fn2:9', + type: 'CALLS', + confidence: 1, + reason: '', + }); + // TAINT_PATH is a whole-program (Function→Function) edge — NOT streamed. + sink.addRelationship({ + id: 'TAINT_PATH:1', + sourceId: 'Function:a.ts:fn:1', + targetId: 'Function:a.ts:fn2:9', + type: 'TAINT_PATH', + confidence: 0.9, + reason: 'src->sink', + }); + + expect(real.nodeCount).toBe(1); + expect(real.relationshipCount).toBe(2); + expect(real.getNode('Function:a.ts:fn:1')).toBeDefined(); + + const manifest = sink.finalize(); + // No BasicBlock node CSV was created (no BasicBlock nodes were routed). + expect(manifest.nodeFiles.size).toBe(0); + expect(manifest.relsByPair.size).toBe(0); + }); +}); + +describe('PdgEmitSink — byte-identity vs whole-graph emit', () => { + it('streamed CSV line set equals streamAllCSVsToDisk for the same nodes/edges', async () => { + const fp = 'a.ts'; + const nodes = [bbNode(fp, 0, 1), bbNode(fp, 1, 5), bbNode(fp, 2, 9)]; + const edges: GraphRelationship[] = [ + pdgEdge(fp, 0, 1, 'CFG', 'seq'), + pdgEdge(fp, 1, 2, 'CFG', 'cond-true'), + pdgEdge(fp, 0, 2, 'REACHING_DEF', 'x:1:0'), + pdgEdge(fp, 1, 2, 'CDG', 'T'), + pdgEdge(fp, 0, 1, 'POST_DOMINATE', ''), + pdgEdge(fp, 0, 2, 'TAINTED', 'taint'), + pdgEdge(fp, 1, 2, 'SANITIZES', 'clean'), + ]; + + // Whole-graph path: add to a plain graph, run streamAllCSVsToDisk. + const wholeGraph = createKnowledgeGraph(); + for (const n of nodes) wholeGraph.addNode(n); + for (const e of edges) wholeGraph.addRelationship(e); + const wholeDir = path.join(tmpRoot, 'csv'); + await streamAllCSVsToDisk(wholeGraph, path.join(tmpRoot, 'no-such-repo'), wholeDir); + + // Streamed path: route the same set through the sink. + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + for (const n of nodes) sink.addNode(n); + for (const e of edges) sink.addRelationship(e); + const manifest = sink.finalize(); + + // BasicBlock node CSV: identical line set. + expect(await sortedLines(path.join(pdgDir, 'basicblock.csv'))).toEqual( + await sortedLines(path.join(wholeDir, 'basicblock.csv')), + ); + + // PDG edges all route to the BasicBlock|BasicBlock pair file: identical set. + expect(await sortedLines(path.join(pdgDir, 'rel_BasicBlock_BasicBlock.csv'))).toEqual( + await sortedLines(path.join(wholeDir, 'rel_BasicBlock_BasicBlock.csv')), + ); + + // Manifest reports the streamed files + row counts. + expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(nodes.length); + expect(manifest.relsByPair.get('BasicBlock|BasicBlock')?.rows).toBe(edges.length); + }); + + it('emits rows via the shared builder (buildBasicBlockRow)', async () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + const n = bbNode('a.ts', 0, 3); + sink.addNode(n); + sink.finalize(); + const lines = await sortedLines(path.join(pdgDir, 'basicblock.csv')); + // header + one data row; the data row is exactly buildBasicBlockRow(n). + expect(lines).toContain(buildBasicBlockRow(n)); + }); +}); + +describe('PdgEmitSink — bounded retention', () => { + it('flushes incrementally so the graph never holds the PDG layer', async () => { + const real = createKnowledgeGraph(); + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const CHUNK = 2; + const sink = new PdgEmitSink(real, pdgDir, CHUNK); // tiny chunk to force flushes + + // TOTAL is intentionally NOT a multiple of CHUNK so the final partial chunk + // is genuinely still buffered (unflushed) at the mid-stream read. With a + // multiple (e.g. 50 % 2 === 0) the last addRow's flush would have written + // every row and the "mid-stream" assertion would prove nothing (#2202 + // review #7). + const TOTAL = 51; + const REMAINDER = TOTAL % CHUNK; // 1 — must be non-zero + expect(REMAINDER).toBeGreaterThan(0); + for (let i = 0; i < TOTAL; i++) sink.addNode(bbNode('a.ts', i, i)); + + // Mid-stream (before finalize): exactly the whole flushed chunks are on + // disk; the partial last chunk (REMAINDER rows) is still buffered in memory, + // proving the writer streams to the OS and never buffers the whole layer. + const midText = fs.readFileSync(path.join(pdgDir, 'basicblock.csv'), 'utf8'); + const midDataRows = midText.split('\n').filter((l) => l.length > 0).length - 1; // minus header + expect(midDataRows).toBe(TOTAL - REMAINDER); // 50 flushed, 1 still buffered + expect(TOTAL - midDataRows).toBe(REMAINDER); // exactly the unflushed remainder + expect(TOTAL - midDataRows).toBeLessThanOrEqual(CHUNK); // unflushed is bounded by one chunk + + // The in-memory graph never received a single BasicBlock. + expect(real.nodeCount).toBe(0); + + const manifest = sink.finalize(); + expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(TOTAL); + const finalRows = (await sortedLines(path.join(pdgDir, 'basicblock.csv'))).length - 1; + expect(finalRows).toBe(TOTAL); // finalize flushed the buffered remainder + }); + + it('finalize twice throws', () => { + const sink = new PdgEmitSink(createKnowledgeGraph(), path.join(tmpRoot, 'pdg-csv')); + sink.finalize(); + expect(() => sink.finalize()).toThrow(/twice/); + }); +}); + +describe('PdgEmitSink — pass-through contract (dedup is the caller’s)', () => { + // The sink does NOT dedup by id (that would retain every id → O(total ids) + // memory, undermining the O(chunk) bound). Cross-pass dedup is done upstream, + // per file, in run.ts (a file imported by two language passes is emitted + // once). The sink is a faithful pass-through: it writes every id it is given + // and must not be fed duplicates. See #2202 finding #1 + the run-loop / Vue+TS + // integration coverage for the cross-pass dedup itself. + it('writes every BasicBlock it is given (no id dedup in the sink)', async () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + const n = bbNode('a.ts', 0, 1); + sink.addNode(n); + sink.addNode(n); // sink does not dedup — both rows are written + const manifest = sink.finalize(); + expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(2); + expect((await sortedLines(path.join(pdgDir, 'basicblock.csv'))).length - 1).toBe(2); + }); + + it('writes every PDG edge it is given (no id dedup in the sink)', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + const e = pdgEdge('a.ts', 0, 1, 'CFG', 'seq'); + sink.addRelationship(e); + sink.addRelationship(e); + const manifest = sink.finalize(); + expect(manifest.relsByPair.get('BasicBlock|BasicBlock')?.rows).toBe(2); + }); + + it('skips a PDG edge whose endpoint label is not a node table', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + // sourceId prefix "Bogus" is not in NODE_TABLES → skipped (mirrors RelPairRouter). + sink.addRelationship({ + id: 'CFG:bogus', + sourceId: 'Bogus:a.ts:0', + targetId: 'BasicBlock:a.ts:1:0:1', + type: 'CFG', + confidence: 1, + reason: 'seq', + }); + const manifest = sink.finalize(); + expect(manifest.relsByPair.size).toBe(0); + }); +}); + +describe('PdgEmitSink — IO failure poisoning (#2202 review #4/#6)', () => { + // A streamed-write failure is an IO fault, not the CFG-logic error the emit + // loop's per-file try/catch is built to swallow. The sink poisons the failing + // writer (or records an open failure) so finalize fails loudly instead of + // returning a truncated manifest that the bulk COPY would silently load. + + afterEach(() => { + vi.restoreAllMocks(); + }); + + it('poisons the writer on a mid-stream write failure (disk-full) → finalize throws', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1); // flush every row + + // The writer opens fine (real openSync); the flush's writeSync fails. + const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => { + throw new Error('ENOSPC: no space left on device'); + }); + // The throw propagates to the immediate caller (the emit loop, which would + // swallow it as a per-file CFG error) — that is exactly why finalize must + // re-check poison below. + expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/ENOSPC/); + spy.mockRestore(); + + expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/); + }); + + it('records an openSync failure (EMFILE) → finalize throws even if the caller swallowed it', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); // ctor mkdir/rm run before the spy + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir); + + const spy = vi.spyOn(fs, 'openSync').mockImplementation(() => { + throw new Error('EMFILE: too many open files'); + }); + expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/EMFILE/); + spy.mockRestore(); + + expect(() => sink.finalize()).toThrow(/IO error|EMFILE/); + }); + + it('surfaces a final-flush IO failure from finalize (rows buffered, never mid-flushed)', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1000); // big chunk → no mid flush + sink.addNode(bbNode('a.ts', 0, 1)); // buffered only (1 < 1000) + + // The only writeSync happens during the final flush inside finalize → close. + const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => { + throw new Error('ENOSPC: disk full at close'); + }); + // close() never throws (it records poison); finalize reports it. + expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/); + spy.mockRestore(); + }); + + it('a poisoned writer stops accepting rows (no unbounded buffering on a dead fd)', () => { + const pdgDir = path.join(tmpRoot, 'pdg-csv'); + const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1); + + const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => { + throw new Error('ENOSPC'); + }); + expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/ENOSPC/); // poisons the writer + // Subsequent rows are dropped silently at the writer (it is dead) — they do + // not re-throw and do not accumulate; the run still fails at finalize. + expect(() => sink.addNode(bbNode('a.ts', 1, 2))).not.toThrow(); + expect(() => sink.addNode(bbNode('a.ts', 2, 3))).not.toThrow(); + spy.mockRestore(); + + expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/); + }); +}); diff --git a/gitnexus/test/unit/stream-pdg-emit-config.test.ts b/gitnexus/test/unit/stream-pdg-emit-config.test.ts new file mode 100644 index 000000000..9d45e8f8a --- /dev/null +++ b/gitnexus/test/unit/stream-pdg-emit-config.test.ts @@ -0,0 +1,99 @@ +/** + * Streaming PDG-emit config gating (issue #2202 U3). + * + * Verifies: + * - `resolveStreamPdgEmit` engages only when pdg + full-rebuild (force) + an + * enable signal (explicit option OR GITNEXUS_STREAM_PDG_EMIT) all hold; + * - `resolvePdgEmitChunkSize` prefers the explicit option, falls back to + * GITNEXUS_PDG_EMIT_CHUNK_SIZE, else undefined; + * - the memory-only streaming knobs are NOT stamped into RepoMeta.pdg, so + * changing them never trips `pdgModeMismatch` (would force needless full + * writebacks otherwise). + */ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import { + resolveStreamPdgEmit, + resolvePdgEmitChunkSize, + pdgModeMismatch, + resolvePdgConfig, +} from '../../src/core/run-analyze.js'; + +afterEach(() => { + vi.unstubAllEnvs(); +}); + +describe('resolveStreamPdgEmit — gating', () => { + it('engages only with pdg + force + an enable signal', () => { + // explicit option + expect(resolveStreamPdgEmit({ pdg: true, force: true, streamPdgEmit: true })).toBe(true); + // env toggle + vi.stubEnv('GITNEXUS_STREAM_PDG_EMIT', '1'); + expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(true); + }); + + it('does NOT engage without --force (incremental writeback reads BasicBlocks back)', () => { + expect(resolveStreamPdgEmit({ pdg: true, force: false, streamPdgEmit: true })).toBe(false); + expect(resolveStreamPdgEmit({ pdg: true, streamPdgEmit: true })).toBe(false); + }); + + it('does NOT engage without --pdg (nothing to stream)', () => { + expect(resolveStreamPdgEmit({ pdg: false, force: true, streamPdgEmit: true })).toBe(false); + }); + + it('does NOT engage with no enable signal', () => { + expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(false); + vi.stubEnv('GITNEXUS_STREAM_PDG_EMIT', '0'); + expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(false); + }); +}); + +describe('resolvePdgEmitChunkSize', () => { + it('prefers the explicit option over the env var', () => { + vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '128'); + expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 999 })).toBe(999); + }); + + it('falls back to the env var, then to undefined', () => { + vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '128'); + expect(resolvePdgEmitChunkSize({})).toBe(128); + vi.unstubAllEnvs(); + expect(resolvePdgEmitChunkSize({})).toBeUndefined(); + }); + + it('rejects non-positive / non-integer env values (falls back to undefined)', () => { + for (const bad of ['0', '-5', 'abc', '1.5', '']) { + vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', bad); + expect(resolvePdgEmitChunkSize({})).toBeUndefined(); + } + }); + + it('rejects an explicit non-positive option (0/negative is not nullish — would defeat buffering)', () => { + expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 0 })).toBeUndefined(); + expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: -10 })).toBeUndefined(); + // ...but an explicit 0 still falls back to a valid env value when present. + vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '256'); + expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 0 })).toBe(256); + }); +}); + +describe('streaming knobs are NOT emit-affecting (no pdgModeMismatch)', () => { + it('changing streamPdgEmit / pdgEmitChunkSize does not trip pdgModeMismatch', () => { + const base = { pdg: true as const }; + const recorded = resolvePdgConfig(base); + // Same pdg config, but with the streaming knobs flipped — must NOT mismatch. + expect( + pdgModeMismatch(recorded, { + ...base, + streamPdgEmit: true, + pdgEmitChunkSize: 64, + } as Parameters[1]), + ).toBe(false); + }); + + it('streaming knobs are absent from the resolved RepoMeta.pdg stamp', () => { + const stamp = resolvePdgConfig({ pdg: true }); + expect(stamp).toBeDefined(); + expect(stamp).not.toHaveProperty('streamPdgEmit'); + expect(stamp).not.toHaveProperty('pdgEmitChunkSize'); + }); +}); From 7d12ea8fd9ded851e8f3458150e33eeaf4d81796 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 16 Jun 2026 05:49:02 +0100 Subject: [PATCH 10/26] feat(analyze): private GitHub repos via PAT + Azure DevOps Server support (#2076, #2210) (#2223) --- .env.example | 6 + .gitleaks.toml | 11 +- .../src/components/AnalyzeOnboarding.tsx | 6 +- gitnexus-web/src/components/RepoAnalyzer.tsx | 140 ++++++++- gitnexus-web/src/lib/lucide-icons.tsx | 35 +++ gitnexus-web/src/locales/en/errors.json | 1 + gitnexus-web/src/locales/en/onboarding.json | 8 +- gitnexus-web/src/locales/zh-CN/errors.json | 1 + .../src/locales/zh-CN/onboarding.json | 8 +- gitnexus-web/src/services/backend-client.ts | 1 + gitnexus/.env.example | 10 + gitnexus/src/server/api.ts | 94 +++++- gitnexus/src/server/git-clone.ts | 223 +++++++++++++- .../server-analyze-token-validation.test.ts | 197 +++++++++++++ gitnexus/test/unit/api-analyze-token.test.ts | 75 +++++ gitnexus/test/unit/git-clone.test.ts | 278 +++++++++++++++++- 16 files changed, 1062 insertions(+), 32 deletions(-) create mode 100644 gitnexus/test/integration/server-analyze-token-validation.test.ts create mode 100644 gitnexus/test/unit/api-analyze-token.test.ts diff --git a/.env.example b/.env.example index 8af9dee79..445ae37a7 100644 --- a/.env.example +++ b/.env.example @@ -17,3 +17,9 @@ WEB_HOST_PORT=4173 # Optional read-only mount, exposed to the server as /workspace. # Override with the directory that contains the repos you want to index. WORKSPACE_DIR=./ + +# Azure DevOps Server Integration (passed to the server container) +# Prefer https:// — the PAT rides in an Authorization header, so cleartext +# http:// exposes it on the wire (still supported for internal-only instances). +# AZURE_DEVOPS_URL=https://azuredevops.example.com +# AZURE_DEVOPS_PAT=your-pat-here diff --git a/.gitleaks.toml b/.gitleaks.toml index 769b7eee9..cecf77d5a 100644 --- a/.gitleaks.toml +++ b/.gitleaks.toml @@ -3,10 +3,17 @@ title = "GitNexus" [extend] useDefault = true -# Fake embedding API keys in unit tests (current probe + historical placeholder). +# Fake credentials in unit tests — none are real secrets: +# - embedding API keys in the http-embedder tests (regexes below) +# - synthetic GitHub PAT fixtures in the git-clone PAT-injection tests +# (e.g. ghp_secret123, ghp_uniqueRawSecret_98765) — allowlisted by path +# so the exception is bounded to that one test file. [allowlist] -description = "fake embedding API keys in http-embedder unit tests" +description = "fake credentials in unit tests (no real secrets)" regexes = [ '''secret-key-12345''', '''test-api-key-redaction-check''', ] +paths = [ + '''gitnexus/test/unit/git-clone\.test\.ts''', +] diff --git a/gitnexus-web/src/components/AnalyzeOnboarding.tsx b/gitnexus-web/src/components/AnalyzeOnboarding.tsx index 53260e388..b970ee842 100644 --- a/gitnexus-web/src/components/AnalyzeOnboarding.tsx +++ b/gitnexus-web/src/components/AnalyzeOnboarding.tsx @@ -3,7 +3,7 @@ * * The "empty state" card rendered inside DropZone's Crossfade when the server * is connected but zero repos are indexed. Replaces the generic error message - * with a first-class GitHub URL input flow. + * with a first-class repository URL input flow. * * Rendering context: * DropZone (Crossfade, phase="analyze") @@ -15,7 +15,7 @@ * the app to the graph explorer. */ -import { Sparkles, Github } from '@/lib/lucide-icons'; +import { Sparkles, GitBranch } from '@/lib/lucide-icons'; import { RepoAnalyzer } from './RepoAnalyzer'; import { useTranslation } from 'react-i18next'; @@ -46,7 +46,7 @@ export const AnalyzeOnboarding = ({ onComplete }: AnalyzeOnboardingProps) => { {/* Icon */}
- +

diff --git a/gitnexus-web/src/components/RepoAnalyzer.tsx b/gitnexus-web/src/components/RepoAnalyzer.tsx index 613284830..fd08d197d 100644 --- a/gitnexus-web/src/components/RepoAnalyzer.tsx +++ b/gitnexus-web/src/components/RepoAnalyzer.tsx @@ -10,12 +10,14 @@ import { useState, useRef, useEffect, useId } from 'react'; import { Github, Gitlab, + AzureDevops, FolderOpen, Loader2, Check, ArrowRight, AlertCircle, Sparkles, + Key, } from '@/lib/lucide-icons'; import { startAnalyze, @@ -30,10 +32,14 @@ import { useTranslation } from 'react-i18next'; // ── Helpers ────────────────────────────────────────────────────────────────── -type InputMode = 'github' | 'gitlab' | 'local'; +type InputMode = 'github' | 'gitlab' | 'azure' | 'local'; const GITHUB_RE = /^https?:\/\/(www\.)?github\.com\/[^/\s]+\/[^/\s]+/i; const GITLAB_RE = /^https?:\/\/[^/\s]+\/[^/\s]+\/[^/\s]+(\/.*)?$/i; +// One-or-more path segments before `/_git/`, so the legacy single-project +// cloud form (myorg.visualstudio.com/project/_git/repo) is accepted too — +// the backend already supports it (isAzureDevOpsUrl / extractRepoName). +const AZURE_RE = /^https?:\/\/[^/\s]+\/(?:[^/\s]+\/)+_git\/[^/\s]+/i; const IS_WINDOWS = navigator.userAgent.toLowerCase().includes('win'); function isValidGithubUrl(value: string): boolean { @@ -44,6 +50,10 @@ function isValidGitlabUrl(value: string): boolean { return GITLAB_RE.test(value.trim()); } +function isValidAzureUrl(value: string): boolean { + return AZURE_RE.test(value.trim()); +} + // ── Mode tabs ──────────────────────────────────────────────────────────────── function ModeTabs({ mode, onChange }: { mode: InputMode; onChange: (m: InputMode) => void }) { @@ -81,6 +91,19 @@ function ModeTabs({ mode, onChange }: { mode: InputMode; onChange: (m: InputMode {t('repoAnalyzer.gitlabUrl')} +