diff --git a/.github/workflows/impact-pdg-mutation-report.yml b/.github/workflows/impact-pdg-mutation-report.yml new file mode 100644 index 000000000..c4bb4fd66 --- /dev/null +++ b/.github/workflows/impact-pdg-mutation-report.yml @@ -0,0 +1,71 @@ +name: Impact PDG Mutation Report + +# Off-the-fast-path mutation oracle for the PDG-backed `impact` mode. +# +# The `--mutation` oracle (bench/impact-pdg/measure.mjs) is a ~280s dynamic +# value-diff check: it mutates each fixture, re-analyzes with `--pdg`, and scores +# the realized recall of the statement slice against the behavioral diff. It is +# far too slow for the PR critical path, so it runs on a nightly schedule (and on +# demand via workflow_dispatch) and uploads the JSON report as an artifact rather +# than gating merges. +# +# The harness shells out to `gitnexus analyze --pdg`, which spawns workers from +# dist/, so dist must be built first — setup-gitnexus with build: 'true' does +# that (mirrors ci-tests.yml). + +on: + schedule: + - cron: '0 3 * * *' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + mutation-report: + name: impact-pdg mutation oracle + runs-on: ubuntu-latest + timeout-minutes: 25 + permissions: + contents: read + steps: + # persist-credentials: false — this job runs the bench + uploads an + # artifact and never pushes; the default-persisted token in .git/config + # must not be capturable through that upload (zizmor credential-persistence + # / artipacked audit). Mirrors ci-tests.yml. + - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + # setup-gitnexus is the repo's composite install action (Node 22 + npm ci); + # build: 'true' runs `node scripts/build.js` so dist/ exists for the + # analyze workers the mutation harness spawns. + - uses: ./.github/actions/setup-gitnexus + with: + build: 'true' + + - name: Run PDG impact mutation oracle (~280s) + run: node --import tsx bench/impact-pdg/measure.mjs --mutation --json > mutation-report.json + working-directory: gitnexus + + # Regression gate: write a recall summary to the run AND fail if the + # minimum realized recall drops below the (tunable) floor, so a recall + # regression surfaces instead of sitting unread in the artifact. Runs + # before the (always) upload so the artifact is preserved even on a fail. + - name: Gate on mutation recall regression + run: node bench/impact-pdg/gate-mutation-recall.mjs mutation-report.json + working-directory: gitnexus + env: + MUTATION_RECALL_FLOOR: '0.5' + + - name: Upload mutation report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: impact-pdg-mutation-report + path: gitnexus/mutation-report.json + retention-days: 14 diff --git a/gitnexus-shared/src/graph/types.ts b/gitnexus-shared/src/graph/types.ts index 085c27d03..8134ad948 100644 --- a/gitnexus-shared/src/graph/types.ts +++ b/gitnexus-shared/src/graph/types.ts @@ -94,6 +94,9 @@ export type NodeProperties = { middleware?: string[]; // BasicBlock (taint/PDG substrate, issue #2080) — reuses filePath/startLine/endLine. text?: string; + /** BasicBlock: space-joined leaf callee names invoked in the block — the + * statement-precise inter-procedural reach substrate for impact mode. */ + callees?: string; // Extensible [key: string]: unknown; }; @@ -171,7 +174,20 @@ export type RelationshipType = * removing it later is a breaking schema change — and it is deliberately * excluded from `VALID_RELATION_TYPES` so it never enters impact-style * symbol-space traversal (same posture as the taint substrate edges). */ - | 'POST_DOMINATE'; + | 'POST_DOMINATE' + /** Per-callee dependence SUMMARY edge (PDG FU-C): a self-loop on a + * Function/Method/Constructor node carrying that callee's RETURN-VALUE + * ASCENT — which formal-parameter indices flow to the function's return + * value, encoded as a versioned bitset in the relation's existing `reason` + * column (the same single-channel pattern `CFG`/`REACHING_DEF`/`CDG` use, + * since the lone `CodeRelation` table has no dedicated label column). A + * later consumer phase lets an interprocedural slice ascend a callee's + * return effect into the caller continuation. Like the taint substrate + * edges it is an internal PDG-engine edge: deliberately EXCLUDED from + * `VALID_RELATION_TYPES` and the web schema so it never leaks into + * callgraph-style impact/relationship surfaces. Emitted only under `--pdg`; + * a default analyze emits zero. */ + | 'CALL_SUMMARY'; export interface GraphNode { id: string; diff --git a/gitnexus/bench/cfg/baselines.json b/gitnexus/bench/cfg/baselines.json index 975a20426..af5f9a800 100644 --- a/gitnexus/bench/cfg/baselines.json +++ b/gitnexus/bench/cfg/baselines.json @@ -9,57 +9,57 @@ "_note": "#2081 M1 / #2082 M2: ONE function, N coalescing statements (extendBlock text accumulation + per-statement fact harvest). Runs at 2000->8000. M2 REWROTE the old 'output is constant 4 blocks' note: statement facts make disk/heap LINEAR in N (a free gate on the harvest payload); TIME still guards the concat path (array-join ~1.0; a genuine O(n^2) re-join accumulation is ~3.8). M2 adds rd_scaling_budget (measured ~0.74) and disk_bytes_large_max -- an ABSOLUTE ceiling ~1.35x the measured indexed-encoding bytes (969,986 at N=8000, ~121 B/stmt); a named-record encoding regression (~4x facts bytes) blows it. Re-baseline the fingerprint only on an intentional CFG/harvest-shape change (the canon now includes statements+bindings)." }, "many-functions": { - "fingerprint": "d881f60e77f0262bdc1b5c7049aa4acf5071e0eabc536476be293c3a133e626e", + "fingerprint": "3a83212717383c2f5cd3179ed28e28d2387ecc0c5ee17044d0290e07da20b8d7", "scaling_budget": 1.5, "disk_bytes_budget": 1.2, "heap_budget": 1.3, "rd_scaling_budget": 2.0, - "_note": "#2081 M1 / #2082 M2 / #2083 M3 U1: N small branchy functions (collect walk + per-function build + per-function solve). Time ~1.0, disk ~1.01, heap ~1.0, rd ~0.86 (solver is per-function; N functions scale linearly). M3 U1 re-fingerprinted: taint sites join StatementFacts (a()/b() call sites); disk_large 2565641->2721641 (+6.1% measured site-harvest cost at N=2000)." + "_note": "#2081 M1 / #2082 M2 / #2083 M3 U1: N small branchy functions (collect walk + per-function build + per-function solve). Time ~1.0, disk ~1.01, heap ~1.0, rd ~0.86 (solver is per-function; N functions scale linearly). M3 U1 re-fingerprinted: taint sites join StatementFacts (a()/b() call sites); disk_large 2565641->2721641 (+6.1% measured site-harvest cost at N=2000). #2227 U1 re-fingerprinted: SiteRecord.at call-site anchor [line,col] joins the statement-facts canon (resolved-callee-id stack); CFG construction/topology unchanged, only the additive serialized `at` byte drifts the JSON.stringify canon. FU-C re-fingerprinted: BindingEntry.formalIndex joins the binding facts canon (CALL_SUMMARY formal-position keying); CFG topology unchanged." }, "branchy": { - "fingerprint": "936765bba5c3f8fc7058737c48351e03e4e1da7fed448467e8fcc8a0fb7786ce", + "fingerprint": "414367e6b351ca9b3cd5df3d7229c607a6c203e600075b9dbaa55f3258e919d6", "scaling_budget": 1.8, "disk_bytes_budget": 1.2, "heap_budget": 1.3, "rd_scaling_budget": 2.0, - "_note": "#2081 M1 / #2082 M2 / #2083 M3 U1: ONE function, N sequential ifs (block/edge growth in one CFG). Time ~1.1-1.25 (noisiest scenario; budget 1.8 absorbs noise, catches ~4.0 quadratic), disk ~1.03, heap ~1.0, rd ~0.7. M3 U1 re-fingerprinted (s{i}() call sites); disk_large 908964->993854 (+9.3%)." + "_note": "#2081 M1 / #2082 M2 / #2083 M3 U1: ONE function, N sequential ifs (block/edge growth in one CFG). Time ~1.1-1.25 (noisiest scenario; budget 1.8 absorbs noise, catches ~4.0 quadratic), disk ~1.03, heap ~1.0, rd ~0.7. M3 U1 re-fingerprinted (s{i}() call sites); disk_large 908964->993854 (+9.3%). #2227 U1 re-fingerprinted: SiteRecord.at call-site anchor [line,col] joins the statement-facts canon (resolved-callee-id stack); CFG topology unchanged. FU-C re-fingerprinted: BindingEntry.formalIndex joins the binding facts canon (CALL_SUMMARY formal-position keying); CFG topology unchanged." }, "dense-bindings": { - "fingerprint": "e4d7eb3c7e8b3772423af25cef391e0e6b68067b554819e81b543439a487403f", + "fingerprint": "ddb5a3389fa629707960bc1892322c0c2735bbad627829d0685e2da89aeb35fa", "scaling_budget": 1.8, "disk_bytes_budget": 1.2, "heap_budget": 1.3, "rd_scaling_budget": 2.0, - "_note": "#2082 M2 / #2201 SSA: N bindings live across ~N blocks in one loop -- bindings x blocks scale JOINTLY (the solver-lattice stressor). The dense GEN/KILL worklist measured rd ~5.2 normalized here (the OUT spine copy is O(V) per block, quadratic when V scales with B). The #2201 SSA-sparse solver answers each use's reaching set from the def-use graph WITHOUT a per-block dense lattice, dropping rd to ~0.86 (linear; measured 5-23x faster absolute). Budget tightened 10->2: still absorbs noise + catches a regression to the per-item-rescan class (a per-use scan over all defs is O(n^3) here, ratio >=16), but now also catches a fall-back to the dense quadratic. Fingerprint unchanged -- CFG construction is untouched." + "_note": "#2082 M2 / #2201 SSA: N bindings live across ~N blocks in one loop -- bindings x blocks scale JOINTLY (the solver-lattice stressor). The dense GEN/KILL worklist measured rd ~5.2 normalized here (the OUT spine copy is O(V) per block, quadratic when V scales with B). The #2201 SSA-sparse solver answers each use's reaching set from the def-use graph WITHOUT a per-block dense lattice, dropping rd to ~0.86 (linear; measured 5-23x faster absolute). Budget tightened 10->2: still absorbs noise + catches a regression to the per-item-rescan class (a per-use scan over all defs is O(n^3) here, ratio >=16), but now also catches a fall-back to the dense quadratic. FU-C re-fingerprinted: BindingEntry.formalIndex joins the binding facts canon (CALL_SUMMARY formal-position keying); CFG topology unchanged." }, "deep-nest": { - "fingerprint": "c0ca870487abc6ff379304c3162003e9e4f9b44aeb2fc29adfcf8d2179c7613a", + "fingerprint": "f2ffa8305a59122f6b52e643f2272e3be882d35bccf87455520ae7fac0b88de2", "scaling_budget": 1.8, "disk_bytes_budget": 1.2, "rd_scaling_budget": 2.0, "facts_large_min": 150, - "_note": "#2201: N nested loops carrying ONE variable end-to-end (depth 40->160) -- the pathology the dense worklist is superlinear on and whose block-visit total drives it past the blocks×64 ceiling (it would TRUNCATE to empty). rd is measured under the PRODUCTION blocks×64 budget (rdProductionBudget) to prove the ceiling stops firing: the depth-INDEPENDENT SSA solver (phi-nodes capture loop merges statically; no fixpoint iteration) computes the full facts (measured 164 at large) with rd_scaling ~0.68 (linear in depth; measured ~0.57ms at depth 160). facts_large_min tightened 100->150 (#2201 review R7): a partial-truncation regression that still cleared the old floor of 100 (but lost facts of the measured 164) now fails, with ~9% headroom under 164 for noise; the companion rd_all_computed gate also catches any non-'computed' status. rd_scaling_budget 2.0 catches a regression back to superlinear. No heap_budget -- the deep-nest CFG payload is tiny and the retained-heap delta is GC-noise-dominated. Re-baseline the fingerprint only on an intentional CFG/visitor change." + "_note": "#2201: N nested loops carrying ONE variable end-to-end (depth 40->160) -- the pathology the dense worklist is superlinear on and whose block-visit total drives it past the blocks×64 ceiling (it would TRUNCATE to empty). rd is measured under the PRODUCTION blocks×64 budget (rdProductionBudget) to prove the ceiling stops firing: the depth-INDEPENDENT SSA solver (phi-nodes capture loop merges statically; no fixpoint iteration) computes the full facts (measured 164 at large) with rd_scaling ~0.68 (linear in depth; measured ~0.57ms at depth 160). facts_large_min tightened 100->150 (#2201 review R7): a partial-truncation regression that still cleared the old floor of 100 (but lost facts of the measured 164) now fails, with ~9% headroom under 164 for noise; the companion rd_all_computed gate also catches any non-'computed' status. rd_scaling_budget 2.0 catches a regression back to superlinear. No heap_budget -- the deep-nest CFG payload is tiny and the retained-heap delta is GC-noise-dominated. Re-baseline the fingerprint only on an intentional CFG/visitor change. FU-C re-fingerprinted: BindingEntry.formalIndex joins the binding facts canon (CALL_SUMMARY formal-position keying); CFG topology unchanged." }, "wide-merge": { - "fingerprint": "7a66a844ee3994bd930c1e34bad3d7b410a762220e927c94fb3787f38d745280", + "fingerprint": "4135a740376068c3f2dedc4751c6588cbe9ae36026ddc9f2ae9112b188980cda", "scaling_budget": 1.8, "disk_bytes_budget": 1.2, "heap_budget": 1.3, "rd_scaling_budget": 2.0, "facts_large_min": 24000, - "_note": "#2201 review R7: N bindings, EACH assigned in a 3-way branch (a wide multi-operand phi per binding) inside a loop, then all used after the merge. Distinct from dense-bindings (one CHAINED redef per `if`): every binding fans into its OWN wide phi, so this exercises phi-placement + renaming + the reachByScc condensation across MANY independent wide merges. N bindings x constant arms => O(N) facts (measured 26008 at the large size), so the gate is rd_scaling LINEARITY: measured ~1.07 (time 9.3->39.8ms over the 4x size step); budget 2.0 catches a regression to the per-binding-rescan O(N^2) class -- the recurring solver antipattern the reachByScc alias fast path (review R2) guards against. rd is measured under the PRODUCTION blocks×64 budget (rdProductionBudget): all functions report 'computed' (the SSA path does not truncate here), and facts_large_min 24000 (measured 26008, ~7% headroom) + the rd_all_computed gate assert the wide merges compute fully. fp_blocks 82 / fp_edges 112 at FP_SIZE=15. Re-baseline the fingerprint only on an intentional CFG/harvest-shape change." + "_note": "#2201 review R7: N bindings, EACH assigned in a 3-way branch (a wide multi-operand phi per binding) inside a loop, then all used after the merge. Distinct from dense-bindings (one CHAINED redef per `if`): every binding fans into its OWN wide phi, so this exercises phi-placement + renaming + the reachByScc condensation across MANY independent wide merges. N bindings x constant arms => O(N) facts (measured 26008 at the large size), so the gate is rd_scaling LINEARITY: measured ~1.07 (time 9.3->39.8ms over the 4x size step); budget 2.0 catches a regression to the per-binding-rescan O(N^2) class -- the recurring solver antipattern the reachByScc alias fast path (review R2) guards against. rd is measured under the PRODUCTION blocks×64 budget (rdProductionBudget): all functions report 'computed' (the SSA path does not truncate here), and facts_large_min 24000 (measured 26008, ~7% headroom) + the rd_all_computed gate assert the wide merges compute fully. fp_blocks 82 / fp_edges 112 at FP_SIZE=15. Re-baseline the fingerprint only on an intentional CFG/harvest-shape change. #2227 U1 re-fingerprinted: SiteRecord.at call-site anchor [line,col] joins the statement-facts canon (resolved-callee-id stack); CFG topology unchanged. FU-C re-fingerprinted: BindingEntry.formalIndex joins the binding facts canon (CALL_SUMMARY formal-position keying); CFG topology unchanged." }, "fact-fanout": { - "fingerprint": "83a8243a8aff117f69aeecb39d02a483e6cca70439d75f63e433f4e4ac85578f", + "fingerprint": "57fa834df795d8dba99947cdfba2bb86b3f9111d8bf86d38b77c155273ae5c93", "scaling_budget": 1.8, "disk_bytes_budget": 1.2, "heap_budget": 1.3, "rd_scaling_budget": 3.0, "facts_large_max": 16000, - "_note": "#2082 M2 / #2083 M3 U1: N switch-arm defs of one variable + N later uses -- facts are O(defs x uses) BY SPEC, so the gate is BOUNDEDNESS, not linearity: with the production fact limit engaged (DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION=16000) the materialized fact count stays pinned at the limit as N grows (facts_large_max), and rd time stays bounded (measured ~1.4). Losing the maxFacts early-stop shows as facts_large exploding quadratically. M3 U1 re-fingerprinted (u{i}(x) call sites); disk_large 996737->1107627 (+11.1%)." + "_note": "#2082 M2 / #2083 M3 U1: N switch-arm defs of one variable + N later uses -- facts are O(defs x uses) BY SPEC, so the gate is BOUNDEDNESS, not linearity: with the production fact limit engaged (DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION=16000) the materialized fact count stays pinned at the limit as N grows (facts_large_max), and rd time stays bounded (measured ~1.4). Losing the maxFacts early-stop shows as facts_large exploding quadratically. M3 U1 re-fingerprinted (u{i}(x) call sites); disk_large 996737->1107627 (+11.1%). #2227 U1 re-fingerprinted: SiteRecord.at call-site anchor [line,col] joins the statement-facts canon (resolved-callee-id stack); CFG topology unchanged. FU-C re-fingerprinted: BindingEntry.formalIndex joins the binding facts canon (CALL_SUMMARY formal-position keying); CFG topology unchanged." }, "taint-dense": { - "fingerprint": "218a1a0c7e092550c233607c67daa401543a25bf8d3f122899d30cd9c30c3a89", + "fingerprint": "f4570c7ae8fde4b7e63b51b4ea641417f911b4267f1f9f2eafadc520020dbf15", "scaling_budget": 1.5, "disk_bytes_budget": 1.2, "heap_budget": 1.3, @@ -69,14 +69,14 @@ "taint_scaling_budget": 2.0, "taint_reason_bytes_large_max": 198000, "taint_zero_match_budget": 0.5, - "_note": "#2083 M3 U7 (R10): N functions, each with 12 req.body sources + a 4-hop chain + 13 eval sinks (13 deduped findings/fn) at 125->500 fns; the zero-match control (inp.payload/evalish) keeps the identical CFG shape with zero model hits. BOUNDEDNESS pin: kept findings/function == 8 (the scenario cap) at BOTH sizes -- above means the cap was lost, below means detection regressed; total findings grow linearly with N by design. disk_bytes_large_max is the LOAD-BEARING site-harvest absolute ceiling (densest sites of the suite; measured 2335772 at N=500, ceiling ~1.35x). taint_reason_bytes_large_max caps the persisted TAINTED reason bytes (measured 146827 = ~37 B/finding, ceiling ~1.35x; blows on hop-encoding bloat or cap loss). taint_zero_match_budget 0.5 vs measured 0.15: the zero-match pass (match gate only, no solver) must stay a small fraction of the match-dense pass. taint scaling measured ~0.93 (per-function work is N-linear); time/disk/heap/rd ratios all ~1.0." + "_note": "#2083 M3 U7 (R10): N functions, each with 12 req.body sources + a 4-hop chain + 13 eval sinks (13 deduped findings/fn) at 125->500 fns; the zero-match control (inp.payload/evalish) keeps the identical CFG shape with zero model hits. BOUNDEDNESS pin: kept findings/function == 8 (the scenario cap) at BOTH sizes -- above means the cap was lost, below means detection regressed; total findings grow linearly with N by design. disk_bytes_large_max is the LOAD-BEARING site-harvest absolute ceiling (densest sites of the suite; measured 2335772 at N=500, ceiling ~1.35x). taint_reason_bytes_large_max caps the persisted TAINTED reason bytes (measured 146827 = ~37 B/finding, ceiling ~1.35x; blows on hop-encoding bloat or cap loss). taint_zero_match_budget 0.5 vs measured 0.15: the zero-match pass (match gate only, no solver) must stay a small fraction of the match-dense pass. taint scaling measured ~0.93 (per-function work is N-linear); time/disk/heap/rd ratios all ~1.0. #2227 U1 re-fingerprinted: SiteRecord.at call-site anchor [line,col] joins the statement-facts canon (resolved-callee-id stack); CFG topology unchanged, disk stays under the 3,150,000 ceiling (measured 2,429,131). FU-C re-fingerprinted: BindingEntry.formalIndex joins the binding facts canon (CALL_SUMMARY formal-position keying); CFG topology unchanged." }, "go:branchy": { - "fingerprint": "bba6ad5452c64125daa1dec4cf25e5e111692748a4ef30f309c9b0e03b3e5017", + "fingerprint": "9baedb7efb61bf1c9d0d8eb51de653258fff52e991834da98de9fc6057f444df", "scaling_budget": 1.8, "disk_bytes_budget": 1.2, "heap_budget": 1.3, "rd_scaling_budget": 2.0, - "_note": "#2195 U7: the first NON-TS scaling scenario -- the C-family analogue of `branchy`, driven through the Go grammar + Go CFG visitor (lang:'go'). ONE Go function with N sequential `if`s (block/edge growth in a single CFG). The `go:` key namespace keeps it out of the TS baseline keyspace (no collision/re-baseline of a TS scenario). Measured time ~1.08, disk ~1.03, heap ~1.0, rd ~1.06 (budgets mirror the TS `branchy` scenario: scaling 1.8 absorbs single-CFG noise + catches a ~4.0 quadratic). Cross-check: fp_blocks 32 / fp_edges 46 are IDENTICAL to the TS branchy fingerprint shape -- the Go visitor builds the same per-`if` block/edge topology. CFG-only (Go has no registered taint model), so no taint gates. Re-baseline the fingerprint only on an intentional Go CFG/harvest-shape change." + "_note": "#2195 U7: the first NON-TS scaling scenario -- the C-family analogue of `branchy`, driven through the Go grammar + Go CFG visitor (lang:'go'). ONE Go function with N sequential `if`s (block/edge growth in a single CFG). The `go:` key namespace keeps it out of the TS baseline keyspace (no collision/re-baseline of a TS scenario). Measured time ~1.08, disk ~1.03, heap ~1.0, rd ~1.06 (budgets mirror the TS `branchy` scenario: scaling 1.8 absorbs single-CFG noise + catches a ~4.0 quadratic). Cross-check: fp_blocks 32 / fp_edges 46 are IDENTICAL to the TS branchy fingerprint shape -- the Go visitor builds the same per-`if` block/edge topology. CFG-only (Go has no registered taint model), so no taint gates. Re-baseline the fingerprint only on an intentional Go CFG/harvest-shape change. #2227 U1 re-fingerprinted: SiteRecord.at call-site anchor [line,col] joins the statement-facts canon (resolved-callee-id stack); Go CFG topology unchanged." } } diff --git a/gitnexus/bench/emit-persistence/baselines-streaming.json b/gitnexus/bench/emit-persistence/baselines-streaming.json index 62e1c0725..67d0ad324 100644 --- a/gitnexus/bench/emit-persistence/baselines-streaming.json +++ b/gitnexus/bench/emit-persistence/baselines-streaming.json @@ -1,4 +1,4 @@ { - "fingerprint": "386f432c74f4992455055d8891dbe6c873ea95afa60ef4be023a21aed7b4bcb1", + "fingerprint": "381de8dede253140953775c290bf9a25f82ebf3cc8ecb5250ae06a63f648b089", "_note": "Byte-identity + bounded-retention gate for streaming/chunked PDG emit (#2202). fingerprint = sha256 of the sorted, header-stripped BasicBlock + PDG-edge data rows of the canonical synthetic set. --check also asserts the streamed PdgEmitSink output is byte-identical to the whole-graph streamAllCSVsToDisk emit (byte_identical_nodes/edges) and that the in-memory graph retains 0 BasicBlocks (resident_basic_blocks === 0, the O(chunk) RSS bound). Regenerate via `node --import tsx bench/emit-persistence/measure-streaming.mjs`." } diff --git a/gitnexus/bench/emit-persistence/baselines.json b/gitnexus/bench/emit-persistence/baselines.json index 93806b252..55307c2fc 100644 --- a/gitnexus/bench/emit-persistence/baselines.json +++ b/gitnexus/bench/emit-persistence/baselines.json @@ -1,5 +1,5 @@ { - "fingerprint": "1b9dd0b783899b47067c36511d241860f291ac736e57682b0ece14148e3958ff", + "fingerprint": "4cc418ea87b6d20a68b5c1139f35d81820b715c63de0ec812e73e2135f5b00b1", "scaling_budget": 1.8, "max_ms_large": 1000, "_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted — binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." diff --git a/gitnexus/bench/impact-pdg/README.md b/gitnexus/bench/impact-pdg/README.md new file mode 100644 index 000000000..845debfa2 --- /dev/null +++ b/gitnexus/bench/impact-pdg/README.md @@ -0,0 +1,798 @@ +# `bench/impact-pdg` — PDG-vs-call-graph impact accuracy harness + +> **STATUS: LIVE (U7, statement-anchored rework).** This directory holds the +> curated ground-truth fixture corpus **and** the measurement harness +> (`measure.mjs`, `metrics.mjs`, `baselines.json`). Run it with +> `node --import tsx bench/impact-pdg/measure.mjs` (build `dist/` first — see *How +> to run*). The harness drives both `impact` engines over the fixtures — PDG +> **seeded on the criterion's statement line** so it returns the dependence slice +> — prints a stratified P/R/F1 table + a plain-language decision recommendation, +> and gates regressions with `--check`. It now also prints an additive **unified +> impact axes** table that keeps line-level and symbol-level truth separate while +> comparing `callgraph`, unified `pdg`, and the evaluation-only +> `composed-current` control baseline. The measured native result remains: +> **PDG is precise at intra-procedural statement granularity (exact on the intra +> AND mixed fixtures — F1 = 1.000; FU-B-2 made the slice statement-granular, closing +> the block-coalescing recall caveat, and the U2 value-diff oracle now agrees on the +> intra stratum); call-graph remains the comparator for inter-procedural symbol +> granularity; unified PDG must match that composed baseline before any +> default-switch decision.** + +## What this measures + +`impact` has two engines that answer **different questions at different +granularities**: + +- `mode: 'callgraph'` (the default) — inter-procedural BFS over symbol→symbol + edges. It answers *"what other symbols depend on / are called by this one?"* at + **symbol granularity**, scored against `inter_AIS`. +- `mode: 'pdg'` (opt-in) — the unified PDG-facing result. Its local + statement slice comes from the persisted CDG + REACHING_DEF Program Dependence + Graph. Seeded with `line: N` (`impact({mode:'pdg', line:N})`), it returns + `affectedStatements: {line, filePath, text}[]` — the dependent **statements** of + the changed line N — and also attaches inter-procedural symbol reach in + `interproceduralByDepth`/`byDepth` for the same target. The native PDG row is still scored against + `intra_AIS`; the unified axes score its statement and symbol outputs together. + +They measure **different scopes**, so the harness scores each at its native +granularity against its native ground truth and reports both side by side. The +"which is more accurate?" question gets an honest, per-scope answer rather than a +single blended number — and the answer is *they answer different questions; +neither strictly dominates*. + +## Unified impact axes + +The harness also reports a separate unified comparison that is designed for the +current architecture question: *does unified `mode:'pdg'` match the composition of today's engines?* This report is additive. It does not +replace the native table above, and it does not change `baselines.json` gating. + +Unified AIS has two namespaces: + +- `statement::` for intra-procedural line truth from `intra_AIS` +- `symbol:@` for inter-procedural symbol truth from `inter_AIS` + +Each engine is adapted onto those axes without lossy projection: + +- `callgraph` contributes only the `symbol` axis. +- `pdg` contributes the `statement` axis from `affectedStatements` and the + `symbol` axis from its unified `interproceduralByDepth`/`byDepth` inter-procedural reach. +- `composed-current` remains an evaluation-only control row that unions standalone + callgraph symbols with PDG statements. + +The report intentionally has no single blended unified F1. `pdg` is now judged +axis-by-axis against `composed-current` so line precision cannot hide +inter-symbol misses, and symbol recall cannot hide statement-level blindness. The +control row is a recall baseline, not a perfection claim: PDG can still +contribute intra-line noise on pure-inter fixtures, so default-switch decisions +should require matching recall while reducing or bounding FPIS. + +> **A note on `line`.** A whole-symbol PDG slice (no `line`) is empty by design: +> intra-procedural dependence stays inside the function, so every reachable block +> is already part of the whole-symbol seed. The useful PDG mode is the +> **statement-anchored** one — seed the criterion's changed statement and read +> the dependent statements back. This is the central change the U7 *rework* +> measures; the earlier "PDG is empty / callgraph wins" verdict was an artifact of +> the whole-symbol seed, now replaced. + +## Runtime result contract + +`impact({mode:'pdg', line:N})` success results carry a target envelope +(`id`, `name`, `type`, `filePath`), `risk: 'UNKNOWN'`, `affectedStatements`, +`affectedStatementCount`, and callgraph-compatible parity fields (`byDepth`, +`byDepthCounts`, `summary`, `affected_processes`, `affected_modules`). +`affectedStatements` is the statement-level PDG slice; `interproceduralByDepth` is the explicit cross-function reach; `byDepth` remains the +compatibility symbol bucket attached by unified PDG mode. + +Degraded PDG results are explicit, not empty successes. `no-layer`, +`sub-layer-missing`, and `unknown` responses keep `mode:'pdg'`, target metadata +when the target resolves, `risk:'UNKNOWN'`, a remediation note, and empty parity +fields. Truncation is also explicit: when both depth and per-step limit bounds +fire, `truncatedByReasons` reports both causes. + +Deferred architecture remains out of scope for this harness: explicit +`Function|Method -> BasicBlock` containment (`CONTAINS_BLOCK`), inter-procedural +summary edges / realizable call-return paths, mutation-derived AIS, and a hybrid +callgraph+PDG impact mode are follow-up features, not assumptions of the current +statement-level benchmark. + +## The corpus + +Each case is a tiny self-contained TypeScript source repo plus a +`ground-truth.json`. TypeScript is used throughout because it has the most +mature CFG/PDG support in this codebase. + +`line` is the `criterion.line` — the statement the PDG slice seeds on. + +| Case | Locus | line | Shape | +|---|---|---|---| +| `intra-dataflow-accumulator` | intra | 8 | loop-carried accumulator def→use (downstream) | +| `intra-dataflow-chain` | intra | 7 | straight-line def→use chain (downstream) | +| `intra-dataflow-reassign` | intra | 9 | reaching defs of a use (upstream, RD-reverse) | +| `intra-control-guard` | intra | 7 | guard-clause control dependence (downstream, CDG-forward) | +| `intra-control-branch` | intra | 7 | if/else-if/else arm control dependence (downstream) | +| `intra-control-loop` | intra | 11 | nested loop+if controllers of a stmt (upstream, CDG-reverse) | +| `inter-dispatcher-thin` | inter | 23 | branch router → 3 handlers (intra slice = routing returns, empty intra_AIS) | +| `inter-facade-delegate` | inter | 21 | guarded sequential delegation chain (empty intra_AIS) | +| `inter-pipeline-stages` | inter | 20 | straight pipeline driver → 3 stages (empty intra_AIS) | +| `mixed-validate-then-call` | mixed | 13 | guard-dominated intra dependence + 1 callee | +| `mixed-compute-and-emit` | mixed | 12 | data-flow-dominated intra dependence + 1 callee | +| `mixed-guarded-dispatch` | mixed | 15 | control+data intra dependence + 2 callees | +| `nobody-interface-excluded` | n/a | — | no-body symbols (KTD6); **excluded** from PDG scoring | + +**Minimum corpus floor (KTD9/F3):** ≥ 3 cases per locus stratum, ≥ 12 total +measurable cases. Current: intra = 7, inter = 3, mixed = 3 → 13 measurable +(+1 excluded no-body case). Below this floor the U7 harness must print +"underpowered — directional only" instead of a verdict. + +## Annotation schema (`ground-truth.json`) + +| Field | Type | Meaning | +|---|---|---| +| `schemaVersion` | int | schema version (currently `1`) | +| `criterion` | `{ name, filePath, direction, line?, marker?, pdgEdgeKinds? }` | the changed symbol — the seed for "what is affected if I change this". `direction` ∈ `downstream` \| `upstream`. **`line`** is the **1-based source line of the statement being changed** — the seed of the statement-anchored PDG slice (`impact({mode:'pdg', line})`). It is chosen from **source semantics** (the def/criterion whose change propagates to the `intra_AIS` lines), *not* by running the traversal (KTD9 annotation-circularity guard), then reconciled against the live traversal in the harness's Step 0. `marker` is a substring unique to the criterion function's body (appears in one of its `BasicBlock.text` fragments); the smoke test uses it to locate the criterion function's blocks deterministically. `pdgEdgeKinds` lists the PDG edge kinds (`REACHING_DEF` \| `CDG`) the criterion function is expected to produce: a pure straight-line data-flow criterion declares only `REACHING_DEF` (no branches → no control dependence), a branching/guard criterion declares both. The smoke test asserts exactly the declared kinds are non-zero on the criterion (so the pure-dataflow archetype isn't forced to carry an artificial branch) and that the criterion produces ≥ 1 PDG edge overall (catching an accidental no-body/zero-edge criterion). `line`, `marker`, and `pdgEdgeKinds` are required for every measurable case; all three are omitted only on `pdgScoring: "exclude"` no-body cases. | +| `intra_AIS` | `AisEntry[]` | symbols/**lines** truly affected WITHIN the same function (the scope where PDG mode is defined). Annotated at **symbol/line granularity, never block-id** (block ids carry fragile `fnLine:fnCol:idx`). | +| `inter_AIS` | `AisEntry[]` | symbols truly affected ACROSS function boundaries (the scope where call-graph mode is defined and intra-procedural PDG is zero-by-design). | +| `locus` | `'intra' \| 'inter' \| 'mixed' \| 'n/a'` | the dominant impact locus; `n/a` only for excluded no-body cases. | +| `pdgScoring` | `'exclude'` (optional) | present (= `"exclude"`) only on no-body cases U7 must drop from PDG denominators. | +| `provenance` | `'manual' \| 'mutation'` | how the AIS was derived. **v1 is `manual` only** — the mutation track (perturb a statement, diff the changed outcomes) needs a fixture-runner + value-diff harness that does not exist yet, so it is deferred. The field stays for forward-compatibility. | +| `analyzerVersion` | string | pinned analyzer version marker (currently the `package.json` version) so ground truth versions against the analyzer. | +| `rationale` | string | prose — WHY each AIS element is in or out. This is what makes manual annotation defensible (SLICEBENCH generate-then-verify discipline). | + +`AisEntry` = `{ symbol, filePath, line?, note? }`. `line` is 1-based and present +for intra entries (which are statement-granular); inter entries name a whole +symbol and omit `line`. + +`intra_AIS` and `inter_AIS` are **disjoint** for every case (an intra entry is a +line within the criterion function; an inter entry is a different symbol). + +## Validity threats (the two that dominate — KTD9) + +1. **Ground-truth incompleteness.** A hand-annotated handful of fixtures yields + *point estimates* over a tiny, self-admittedly incomplete corpus. One + mis-annotation can swing F1 by a large fraction, so U7 reports findings as a + **direction**, not a headline decimal, until the corpus grows / the mutation + track lands. +2. **Annotation circularity.** PDG's `intra_AIS` risks being reconciled against + the PDG traversal's own output. **Mitigation (KTD9 annotation-circularity + guard): these annotations are written from SOURCE SEMANTICS first** — reading + the source and reasoning about def→use / control dependence by hand — and + reconciling against the live traversal is **U7's job (its Step 0), not the + annotation's**. Call-graph gets no such home-field annotation, so the + comparison is not rigged toward PDG. + +## Methodology — CIS / AIS, stratified (KTD9, Arnold–Bohner) + +For each fixture × mode the harness compares the mode's **CIS** (Computed Impact +Set — what it reports as impacted) against the **AIS** (Actual Impact Set — the +curated ground truth), at the mode's **native granularity**, stratified by impact +locus: + +- **precision** = |AIS∩CIS| / |CIS| (over-approximation cost), +- **recall** = |AIS∩CIS| / |AIS| (under-approximation; the *dangerous* miss for + a safety tool), +- **F1** = harmonic mean, +- **FPIS** = CIS − AIS (noise), **FNIS** = AIS − CIS (missed), +- **|CIS|/|AIS|** size ratio. + +**Each engine is scored at its own granularity against its own ground truth:** + +- **PDG → line granularity vs `intra_AIS`.** CIS_pdg is the set of + `affectedStatements` **line** keys (`:`) returned by the + line-seeded slice; AIS is the `intra_AIS` line set. This is the unit at which + PDG is precise — the dependent *statements* of the changed line. +- **Call-graph → symbol granularity vs `inter_AIS`.** CIS is the reported + **symbol** keys (`@`); AIS is the `inter_AIS` symbol set. + This is the unit at which the cross-function blast radius is meaningful. + +**Empty-denominator semantics are explicit, never silently 0/1.** |CIS|=0 ⇒ +precision is `n/a` (no predictions); |AIS|=0 ⇒ recall is `n/a` (no truth in that +scope). A scope with an `n/a` metric is **excluded** from that metric's mean, +never folded in as 0 (the apples-to-oranges trap, R1). The pure scorer lives in +`metrics.mjs`; its arithmetic is pinned by the deterministic unit test +`test/unit/impact-pdg-metric-math.test.ts` (synthetic sets only — no analyze, no +DB, so it stays out of the flaky full-pipeline lane). + +**Stratification.** Each fixture is scored in its **own** locus stratum +(intra/inter/mixed). Within a stratum, the PDG row is line-vs-`intra_AIS` and the +call-graph row is symbol-vs-`inter_AIS`: + +- On an **intra** fixture, `inter_AIS` is empty, so call-graph reports no other + symbol → its row is `n/a` (no cross-function truth). PDG is scored against the + real `intra_AIS`. +- On an **inter** fixture, `intra_AIS` is empty by design, so the PDG line slice + returns only the router's own control-dependent statements — FPIS against the + empty truth (precision 0, recall `n/a`). Call-graph is scored against the real + `inter_AIS`. This is the honest *"PDG is intra-procedural; on a pure-inter + fixture it has no meaningful intra ground truth"* result — **symmetric** to + call-graph's empty intra row. +- On a **mixed** fixture, both rows are real: PDG resolves the intra statement + set, call-graph reaches the callee(s). + +**The native rows still measure different units.** The PDG native row scores +statement reach, while the callgraph native row scores symbol reach. The unified +axes table is where `pdg` is judged as the composed result: statement reach in +`affectedStatements`, inter-symbol reach in `byDepth`. + +## Substrate (the load-bearing mechanism — R8) + +`runPipelineFromRepo` is in-memory and never persists, but `impact` queries a +**persisted** `lbugPath` + a `meta.pdg` stamp; there is no exported `runAnalyze` +(the entrypoint `analyzeCommand` calls `process.exit`, unusable in a loop), and +the test-suite `vi.mock` bridge is vitest-only. So the harness runs **real +analyze via a temp `GITNEXUS_HOME`, mock-free**. Per fixture: + +1. Point `process.env.GITNEXUS_HOME` at a per-run temp dir (honored by + `repo-manager.getGlobalDir()` — it roots the registry; the per-repo DB lands + in `/.gitnexus/`, so fixtures are copied to a temp working dir + first, keeping the source tree clean). +2. **Shell out** to the real CLI as a child process — child-process isolation + sidesteps `process.exit`; real `saveMeta` + `registerRepo` land in the temp + home; parse workers spawn from `dist/` (so the harness needs a built `dist/`): + + ``` + node --import tsx src/cli/index.ts analyze --pdg --skip-git --index-only + ``` +3. `new LocalBackend(); await init()` resolves the fixture via the **real** + registry (the parent process sets `GITNEXUS_HOME` too, so `init()` reads the + temp registry, not `~/.gitnexus`). +4. `callTool('impact', …)` ×2 (the absolute path is a tier-1 path match — no + name collision): once `mode:'callgraph'` (symbol BFS), once `mode:'pdg'` with + `line: criterion.line` so it returns the **statement-anchored slice** + (`affectedStatements`). A whole-symbol PDG slice (no `line`) is empty by + design, so the seed line is load-bearing. +5. Teardown the temp home + copy. + +### Step 0 — fixture AIS validation (gated on the live traversal; circularity) + +Before scoring, the harness reconciles each fixture against the live analyzer +(`metrics.mjs` is annotation-only; Step 0 is the *traversal* reconciliation): + +- the criterion must produce **≥ 1 PDG edge** (an accidental no-body / cap- + truncated criterion has unmeasurable ground truth → excluded, logged); +- the criterion symbol must **not** share `(filePath, startLine)` with another + `Function`/`Method` (one count query) — same-line projection ambiguity (R4) + would reconcile AIS against the wrong symbol's edges → excluded, logged. + +Per the **annotation-circularity guard**, this reconciliation runs *second*: the +`criterion.line` and the AIS were written from source semantics *first* (read the +source, find the def/criterion whose change propagates), and Step 0 only confirms +the fixture is measurable substrate — it never *derives* ground truth from the +traversal. Where a source-derived belief disagreed with the live block-granular +traversal, the **annotation** was corrected (documented in each +`ground-truth.json` rationale), not the metric re-fit: + +- **Direction.** `inter-pipeline-stages`'s AIS named callees while the criterion + was tagged `upstream`; the annotation was corrected to `downstream`. +- **Block coalescing — RESOLVED (FU-B-2, statement-granular).** The CFG coalesces + consecutive straight-line statements into one `BasicBlock`. Before FU-B-2 the + block-granular slice could not pinpoint a coalesced block's *interior* + statements, so `intra-dataflow-chain` (8,9 → inside the line-7 seed block), + `intra-control-guard` (12 → inside the line-11 body block), and + `intra-dataflow-reassign` (8 → inside the line-7 def block) had their `intra_AIS` + interior lines removed as block-granularity artifacts. **FU-B-2 makes the intra + slice statement-granular**: each persisted `REACHING_DEF` edge now carries its + def/use *source lines* (a compact versioned annotation on `reason`), and the + projection walks the self-edge def→use line chain forward from the criterion + (and through every reached coalesced block) to recover those interior + statements. So the three fixtures were re-reconciled UP — chain {10}→{8,9,10}, + guard {9,11,13}→{9,11,12,13}, reassign {6,7}→{6,7,8} — restoring the original + source-derived belief the prior block-granularity reconciliation had + under-counted. The annotation fingerprint moved deliberately; the U2 value-diff + oracle had already proved chain's {8,9} independently, so this is a justified + ground-truth correction, not a metric re-fit. +- **Under-counted dependencies.** The combined CDG+REACHING_DEF slice reaches more + than a control-only or single-step reading: `intra-control-branch` (+line 10, + the nested `else if` predicate, control-dependent on the outer branch), + `intra-control-loop` (+lines 6,7, the param block and `count` init reaching the + increment), and `intra-dataflow-reassign` (+line 6, the param def of `a`) gained + lines the original annotation missed. + +After reconciliation, the line-seeded slice reproduces each corrected `intra_AIS` +exactly (FPIS = FNIS = 0) on all 7 intra fixtures AND all 3 mixed fixtures (the +FU-A intra-tag scopes the intra axis to the criterion's own function, so the U1 +cross-function callee lines no longer count as intra FPIS). The U2 value-diff +oracle now agrees with the static slice at statement granularity on the intra +stratum (chain's {8,9} are in the slice). Call-graph gets no such home-field +annotation, so the comparison is not rigged toward PDG. + +## Measured results (analyzer 1.6.7, 13 measurable + 2 excluded; post-U1 + U2 + FU-B-2) + +Each engine scored at its **native granularity** against its **native ground +truth** — PDG at line vs `intra_AIS`, call-graph at symbol vs `inter_AIS`: + +| Scope | Mode | Granularity | P | R | F1 | \|CIS\|/\|AIS\| | FPIS | FNIS | n | +|---|---|---|---|---|---|---|---|---|---| +| intra | callgraph | symbol/inter | n/a | n/a | n/a | n/a | 0 | 0 | 7 | +| intra | **pdg** | **line/intra** | **1.000** | **1.000** | **1.000** | 1.000 | 0 | 0 | 7 | +| inter | **callgraph** | **symbol/inter** | **1.000** | **1.000** | **1.000** | 1.000 | 0 | 0 | 3 | +| inter | pdg | line/intra | 0.000 | n/a | n/a | n/a | 10 | 0 | 3 | +| mixed | **callgraph** | **symbol/inter** | **1.000** | **1.000** | **1.000** | 1.000 | 0 | 0 | 3 | +| mixed | **pdg** | **line/intra** | **1.000** | **1.000** | **1.000** | 1.000 | 0 | 0 | 3 | + +> **Post-FU-B-2 correction.** FU-B-2 makes the intra slice **statement-granular** — +> the persisted `REACHING_DEF` edge carries its def/use source lines, and the +> projection walks the self-edge def→use chain (forward from the criterion, and +> through every reached coalesced block) to recover interior statements. With the +> three coalesced-block fixtures re-reconciled UP (chain {10}→{8,9,10}, guard +> {9,11,13}→{9,11,12,13}, reassign {6,7}→{6,7,8}), **intra/pdg stays F1 = 1.000 +> (FPIS = FNIS = 0)** and the U2 value-diff oracle now AGREES with the slice on the +> intra stratum (the old 0.333 statement-level recall on `intra-dataflow-chain` is +> now 1.000 — the block-coalescing blind spot is closed, not merely matched by a +> blind annotation). **mixed/pdg is now F1 = 1.000** (was 0.468 post-U1): the FU-A +> intra-tag scopes the intra axis to the criterion's own function, so the U1 +> cross-function callee statements live on the inter symbol axis, not as intra +> FPIS. The remaining inter/pdg FPIS = 10 are the router's own control-dependent +> returns scored against an empty `intra_AIS` (by design — see the `n/a`/`0` +> explanation below). + +Read it honestly: + +- **PDG mode is precise at intra-procedural statement granularity — exact on the 7 + intra AND the 3 mixed fixtures.** The line-seeded slice returns *exactly* the + reconciled `intra_AIS` (F1 = 1.000, FPIS = FNIS = 0) on both strata. It precisely + identifies the dependent statements of the changed line (def→use chains, + control-dependent arms, reaching defs); the earlier "empty / no signal" result was + the whole-symbol-seed artifact, and the post-U1 mixed precision dip (0.468) was + closed by the FU-A intra-tag (cross-function callee lines score on the inter axis, + not as intra FPIS). **FU-B-2 closed the block-coalescing blind spot:** the intra + slice is now statement-granular (REACHING_DEF edges carry their def/use source + lines; the projection walks the self-edge def→use chain through each coalesced + block), so the U2 value-diff oracle that previously proved a statement-level recall + of 0.333 on `intra-dataflow-chain` (lines `chain.ts:8,9`) now measures **1.000** — + the slice and the dynamic oracle agree at statement granularity on the intra stratum. +- **Call-graph mode is exact on the cross-function questions.** On all 3 inter + fixtures and all 3 mixed fixtures it recovers every callee — F1 = 1.000. It is + the engine for "what else calls/uses this?". +- **The two `n/a` / `0` cells are by design, not defects.** *intra/call-graph*: a + self-contained function calls no other symbol, so call-graph reports nothing and + `inter_AIS` is empty → no cross-function truth to score (`n/a`). *inter/pdg*: a + pure-inter router has an empty `intra_AIS`, and the line-seeded slice returns the + router's *own* control-dependent routing returns — FPIS against the empty truth + (precision 0, recall `n/a`). These are **symmetric**: each engine is blind to + the other's native scope. The per-case lines surface each statement slice (`pdg + line/intra: …`) and each callee set (`cg symbol/inter: …`), while the unified + table verifies whether `pdg` now carries both axes. + +## Decision recommendation (the verdict — F2) + +> **The two engines answer different questions at different granularities, and +> neither dominates.** +> +> - **`mode:'callgraph'` (the default)** is the correct engine for the +> *inter-procedural* safety question — *"what else depends on / calls this +> symbol?"* It recovers the cross-function callees exactly (inter & mixed F1 = +> 1.0 on this corpus) and carries the cross-function reach the blast radius +> needs. Use it for cross-symbol impact. +> - **`mode:'pdg'` (opt-in, seeded with `line:N`, where `analyze --pdg` persisted +> the layer)** is **precise at intra-procedural *statement* granularity** — +> *"which statements inside this function does changing line N affect?"* On the +> 7 intra fixtures AND the 3 mixed fixtures it reproduces the dependent-statement +> set exactly (intra & mixed PDG F1 = 1.0, FPIS = FNIS = 0): the FU-A intra-tag +> scopes the intra axis to the criterion's own function (cross-function reach goes +> on the inter symbol axis), and FU-B-2 made the slice statement-granular so the U2 +> value-diff oracle now agrees on the intra stratum (the old block-coalescing +> recall caveat — 0.333 on one chain fixture — is closed: recall 1.000). This +> is still a question call-graph **cannot answer at all** (it has no notion of a statement). +> +> `mode:'pdg'` now composes those surfaces in one result: `affectedStatements` +> carries statement-level dependence and `interproceduralByDepth`/`byDepth` carries +> inter-procedural symbols. `mode:'callgraph'` remains the option-driven comparator/default. The +> unified axes table keeps `composed-current` as the control baseline that PDG +> must match or beat before any default-switch decision. +> match or exceed while reducing or bounding FPIS. Reach for the line-seeded +> PDG when you need statement-level dependence *inside* a function; reach for +> call-graph when you need +> *cross-function* reach. The earlier verdict +> ("PDG is empty / call-graph wins") was an artifact of the **whole-symbol** seed +> — a whole-symbol slice has nothing to report because intra-procedural dependence +> never leaves the function. Seeding the changed *statement* is what makes PDG's +> precision measurable, and it measures as exact. + +## Validity threats (the two that dominate — KTD9) + +1. **Ground-truth incompleteness.** A hand-annotated handful of fixtures yields + *point estimates* over a tiny, self-admittedly incomplete corpus. One + mis-annotation can swing F1 by a large fraction, so the harness reports + findings as a **direction**, not a headline decimal, and prints an explicit + "underpowered — directional only" banner when the corpus falls below the + floor. +2. **Annotation circularity.** PDG's `intra_AIS` risks being reconciled against + the PDG traversal's own output. **Mitigation:** these annotations are written + from SOURCE SEMANTICS first (U6) — reading the source and reasoning about + def→use / control dependence by hand — and reconciling against the live + traversal is the harness's **Step 0**, run *second*, only to confirm + measurability. Call-graph gets no such home-field annotation, so the + comparison is not rigged toward PDG. + +## Underpowered-corpus rule (F3) + +**Minimum corpus floor: ≥ 3 measurable cases per locus stratum, ≥ 12 total.** +Current corpus is above the floor (intra 7, inter 3, mixed 3 = 13 +measurable; +1 excluded no-body) — so the harness prints headline decimals. When +the measurable count after exclusions drops below the floor, it instead prints +**"underpowered — directional only"** and reports the DIRECTION ("PDG exact at +intra statement granularity; call-graph exact at inter symbol granularity") +rather than headline decimals — decimal precision (`F1 0.74 vs 0.68`) implies a +confidence a sub-floor corpus cannot support. Even at the floor the F1 = 1.0 +results should be read as *"exact on this small, deliberately-simple corpus"*, not +*"exact in general"* — see the validity threats. + +## Annotation fingerprint + `--check` (two gates, KTD10) + +`--check` runs **two non-byte-identity gates** (an exact-equality gate would go +perpetually red on legitimate accuracy changes): + +1. **One-sided F1 regression band** per mode per scope: fail iff `F1 < band − ε`; + improvements pass freely. `ε` and the per-`(scope,mode)` bands are versioned + in `baselines.json`. The four live bands are **intra/pdg = 1.0**, **mixed/pdg = + 1.0**, **inter/callgraph = 1.0**, **mixed/callgraph = 1.0**. A `null` band + means F1 is genuinely undefined for that cell on this corpus (intra/callgraph + and inter/pdg — see *Measured results*) — the gate skips it. +2. **Order-independent annotation fingerprint** over the curated ground-truth + set (a SHA-256 over a sorted, line-collapsed canonicalization — mirrors the + `bench/cfg/measure.mjs` *technique*, written here, not a literal import). Any + unreviewed edit to a `ground-truth.json` (criterion **including + `criterion.line`**, AIS membership, locus, direction, edge kinds) trips it; a + pure reordering of AIS entries does not. + +**Substrate stability (F5).** Real analyze is the repo's flaky lane, so `--check` +applies **median-of-K** across `GN_IMPACT_PDG_K` runs *before* comparing F1 to +the band, so substrate noise can't trip the metric gate. Default K = 1 (the +fixtures are tiny and deterministic in practice); raise it +(`GN_IMPACT_PDG_K=3`) in a flaky CI lane. + +## Runtime budget + +Each fixture costs **one full `analyze --pdg` child process** (a fresh tree-sitter +parse + CFG/PDG build + persist) plus two in-process `impact` calls (one +call-graph, one line-seeded PDG). On these tiny fixtures that is ≈ +**3–6 s/fixture**, so the full 13-fixture corpus runs in roughly **45–80 s** +wall-clock single-threaded (K = 1). A K-fold `--check` multiplies by K. For a +fast substrate smoke, scope to a subset: +`--only=intra-dataflow-chain,inter-dispatcher-thin,mixed-guarded-dispatch` (or +`GN_IMPACT_PDG_ONLY=…`). Not wired into `npm test` (matches the other benches); +the deterministic metric-math unit test *is* in `npm test`. + +## How to run + +```sh +cd gitnexus +node scripts/build.js # REQUIRED: workers spawn from dist/ +node --import tsx bench/impact-pdg/measure.mjs # print the stratified report + verdict +node --import tsx bench/impact-pdg/measure.mjs --json # machine report (for re-baselining) +node --import tsx bench/impact-pdg/measure.mjs --check # gate against baselines.json (exit non-zero on regression) +node --import tsx bench/impact-pdg/measure.mjs --only=a,b,c # fast subset (substrate smoke) +node --import tsx bench/impact-pdg/real-code.mjs # latency + quality-proxy probe on indexed GitNexus +node --import tsx bench/impact-pdg/real-code.mjs --json --check # machine report + broad real-code gates +node --import tsx bench/impact-pdg/blast-radius.mjs # real-code localization: PDG slice vs whole-function body +node --import tsx bench/impact-pdg/blast-radius.mjs --direction upstream +``` + +### Real-code performance and quality proxy probe + +`real-code.mjs` complements the AIS-backed fixture harness. It runs direct +`LocalBackend.callTool("impact", ...)` calls against an already-indexed real +repository (default `--repo GitNexus`) and measures: + +- callgraph vs PDG median/p95 latency over `--repeat` samples; +- whether unified PDG's inter-procedural symbol reach preserves the callgraph + symbol set for the same target/direction; +- degraded, partial, no-block-at-line, and PDG bridge evidence counts. + +This is a quality proxy, not an accuracy score: a real repo has no curated AIS, +so the probe cannot prove correctness. Use it to catch performance regressions, +degraded indexes, symbol-reach drift, and excessive `unproven-bridge` evidence on +real code. Use `measure.mjs` for the ground-truth precision/recall/F1 gate. + +The default cases are statement-anchored at a CFG **block-start** line. The CFG +coalesces straight-line statements into one `BasicBlock`, so a mid-block anchor +resolves to no block start and degrades to `pdg-no-block-at-line` — honest, but it +then exercises only the symbol axis. The harness still detects and counts that +degradation; the curated anchors avoid it so every case also exercises a real +intra-procedural slice. (This is the same statement-anchoring discipline the +fixture corpus uses, applied to real code.) + +A representative run on the indexed GitNexus tree (~17.5k symbols, PDG layer +persisted via `analyze --pdg` with ~171k `BasicBlock`s) — read it *directionally*, +not as a baseline, since wall-clock latency is host- and noise-dependent: + +- **Symbol reach is preserved exactly.** Unified `mode:'pdg'` reproduces the + `mode:'callgraph'` inter-procedural symbol set on every case — mean and min + recall = precision = **1.000**. This is the load-bearing check: the PDG-facing + result must not silently drop or invent cross-function reach. +- **Each case carries a real statement slice** (`affectedStatements` non-empty, + 2–27 statements here), so the intra axis is genuinely exercised. +- **Latency overhead is modest** — PDG median ≈ **1.2–1.4×** the callgraph median + (callgraph ≈ 90–250 ms/case, PDG ≈ 150–280 ms/case). The first call of a fresh + backend carries a one-time DB-warmup spike the p95 reflects. +- **Bridge evidence is direction-shaped, by design.** Downstream + statement-anchored seeds label most inter-procedural reach `unproven-bridge` + (the symbol's first-hop call site sits in a *different* statement than the + seeded one, so the local slice does not prove the dependence); upstream and + whole-symbol reach is `callgraph-bridge`. So `unprovenBridgeRatio ≈ 0.7` is the + *expected* shape for statement-anchored downstream seeds — a faithful + proven-vs-reachable signal, **not** a regression. +- **No degraded / error / partial / no-block-at-line cases**, and `--check` is + green. Default gates: min symbol recall ≥ 0.95, PDG median ≤ 5000 ms (override + via `GN_REAL_CODE_PDG_MIN_SYMBOL_RECALL` / `GN_REAL_CODE_PDG_MAX_MEDIAN_MS`). + +### Is PDG-mode impact actually better than callgraph-only? (four-axis verdict) + +"Better" is not one thing, so each candidate claim is tested separately and +reported honestly — including where PDG is *not* better. The evidence combines +the AIS-backed fixture gate (`measure.mjs`, which proves *correctness*) with two +real-code probes on the live GitNexus index (`real-code.mjs` and +`blast-radius.mjs`, which measure *magnitude at scale*: 120 functions per +direction, 240 total, plus the 5-case probe). `blast-radius.mjs` anchors each +function on an early-interior block (`floor(M/3)`), a conservative slice-maximizing +choice, and compares the PDG statement slice to the whole function body (`M` +blocks). + +| Claim | Verdict | Evidence | +|---|---|---| +| **Tighter / fewer false alarms** | ✅ confirmed for localization and correctness | *Correctness:* the line-seeded slice equals the curated intra dependence exactly on the 7 **intra** fixtures AND the 3 **mixed** fixtures (F1 = 1.000, FPIS = FNIS = 0): the FU-A intra-tag keeps cross-function reach on the inter axis, and FU-B-2's statement-granular slice closed the block-coalescing blind spot — the U2 value-diff oracle now measures statement-level recall **1.000** on the chain fixture (was 0.333). *Magnitude (RECORDED, not re-run this session):* the slice is a median **0.26** (downstream) / **0.21** (upstream) of the function body; **240/240** functions localized below whole-body — a ~74–79% cut in the intra-procedural inspection set, with no proven dropped dependency. | +| **Catches impact callgraph misses** | ✅ confirmed (new axis) | Callgraph emits *no* statement-level output (unified intra-line CIS = 0, recall 0 on every fixture); PDG recovers every true dependent statement (intra recall = 1.000). PDG answers a def→use / control-dependence question callgraph cannot represent at all. | +| **Finds *more* callers/callees** | ❌ refuted (tie, by design) | Full PDG inter-procedural reach is **identical** to callgraph on 240/240 real functions (0 pdg-only, 0 callgraph-only). PDG bridges inter-procedural reach *through* the call graph, so it never finds reach the call graph misses. | +| **Tighter cross-function reach (statement-precise)** | ✅ confirmed (precision, additive) | `mode:'pdg'` now also exposes `statementPreciseByDepth` — the callees actually invoked from the changed line's dependence slice (`BasicBlock.callees`), dropping symbols only reachable from independent statements. Strictly tighter than callgraph on **52/90** with-slice functions (median proven **1** vs callgraph **2** symbols, median statement-precision **0.67**); the full reach stays available alongside it. `statementPrecision` reports the cut. Upstream seeds have no statement discriminator, so they stay all-proven (callgraph-equal) by design. | +| **Faster / cheaper** | ❌ refuted | PDG carries ~**1.2–1.6×** callgraph latency (the slice query + the slice-callees lookup). It buys precision, not speed. | + +**Headline.** PDG makes `impact` *much* better at the localization/precision +question — *"what exactly does changing **this** statement affect?"* It narrows the +intra-procedural blast radius to roughly a quarter-to-a-third of the function body +with ground-truth-proven correctness, adds a statement-level dependence axis +callgraph has no answer for, and — via the persisted `BasicBlock.callees` substrate +— now also reports a **statement-precise** cross-function reach (only the callees +the changed line actually reaches), strictly tighter than callgraph on roughly half +of with-slice functions. It is deliberately **not** a *wider* or *faster* +cross-function reach: the full callgraph reach is preserved alongside the precise +view, and `mode:'callgraph'` remains the comparator for raw blast radius. The +surfaces compose — that is the point of the unified result, not a default switch. + +Reproduce the verdict: + +```sh +node --import tsx bench/impact-pdg/measure.mjs # correctness (F1 / FPIS / FNIS vs AIS) +node --import tsx bench/impact-pdg/blast-radius.mjs # localization magnitude (downstream) +node --import tsx bench/impact-pdg/blast-radius.mjs --direction upstream +node --import tsx bench/impact-pdg/real-code.mjs # symbol-reach preservation + latency +``` + +### Resolved-symbol-id soundness (the `calleeIds` upgrade) + +The statement-precise cross-function bridge originally matched callgraph-reached +callees to the slice by **leaf name** (`BasicBlock.callees`). That is a heuristic +with two failure modes: same-leaf-name **collision** (two distinct `get`s both +proven — a false positive) and import-alias/rename (call-site leaf ≠ resolved name +— a false negative). The bridge now matches the **resolved callee symbol-id** +(`BasicBlock.calleeIds`, the per-block union of resolved ids joined to each call +site by exact position), which is sound by construction; the leaf-name match +remains the graceful fallback for pre-v3 indexes, blocks with no captured ids, and +truncation-capped blocks. `name-collision.mjs` diffs the two on the same real +slices (`fpEliminated` = collision FPs the id bridge removes; `fnRecovered` = +alias FNs it recovers). + +Realized effect on a random single-statement sample (per language, exact +seed∪reachable slice): + +| Language | repo | fpEliminated | fnRecovered | name-collision ambiguity | +|---|---|---:|---:|---:| +| Java | commons-lang | 2.1% | 0% | 3.9% | +| PHP | monolog | 2.6% | 1.8% | 12.2% | +| C# | commandline | 0% | 4.8% | 0% | +| TS | ky | 0% | 0% | 0% (no regression) | + +Honest reading of these numbers: + +- **The aggregate effect on a *median* edit is modest (≈0–3%).** This matches the + pre-build measurement: realized name-collision concentrates in the small tail of + high-fan-out delegating functions, not the typical single-statement slice (the + per-function reach is usually 1–2 callees, where a same-name collision is + impossible). The win is **soundness**, gated cleanly by the + `intra-overloaded-callee` fixture (id proves exactly the one called overload; + name-match over-attributes both — `measure.mjs --check` Gate 3), not a large + aggregate FP cut. +- **It is bidirectional.** The id key also *recovers* alias/rename false negatives + the name match can never prove (C# 4.8%, PHP 1.8%) — callees invoked under a + name that differs from their resolved symbol name. +- **The id bridge is exactly as precise as GitNexus's call resolver — no more, no + less.** Where the resolver emits a *multi-candidate* set for one ambiguous call + (e.g. `printer.getX()` on a typed field resolving to `getX` on **both** the field + type and the enclosing class), the bridge faithfully proves the whole candidate + set (sound — it never drops a real target). The residual "ambiguity" on the + worst-case tail is therefore the **resolver's** receiver-type precision, not a + name-matching artifact; improving it is a resolver-precision follow-up (sibling + to the C++ overload under-resolution follow-up). + +**Language scope.** The id bridge (and the name bridge) applies wherever the CFG +harvests call sites. As of the call-site-harvesting extension this is **all 12 +supported languages** — the original six (TS/JS, Java, C#, Go, C/C++, PHP) plus +Kotlin, Swift, Dart, Ruby, Rust, and Python, which were migrated from the no-site +def/use accumulator to the shared `CallSiteFactAccumulator` (each verified that its +`SiteRecord.at` anchor matches that language's `@reference.call` resolution anchor +byte-exact, so the resolved-id join lands). Their BasicBlocks now carry `callees` +*and* `calleeIds`. Realized benefit still tracks each language's collision tail and +its call-resolver precision (e.g. Python/Ruby route most calls to stdlib/builtins, +which carry no in-repo id), but the substrate is uniform. The one remaining +language-shaped gap is **C++ overload under-resolution** (a resolver issue, not a +harvesting one — see the C++ caveat above). + +Reproduce (needs a `--pdg` index of the target repo built under schema v3): + +```sh +node --import tsx bench/impact-pdg/name-collision.mjs --repo commons-lang --src 'src/main/java/' +node --import tsx bench/impact-pdg/name-collision.mjs --repo monolog --src 'src/Monolog/' +``` + +### Inter-procedural forward slice (U1 — `calleeIds` descent) + +The `mode:'pdg'` slice was originally **intra-procedural**: the CDG + +REACHING_DEF traversal stayed inside the seeded function, and cross-function +reach was bolted on only through the call-graph bridge. U1 makes the statement +slice itself cross function boundaries: after the intra slice completes (and +before block→symbol projection), a bounded **DOWNSTREAM-only** descent gathers +the slice blocks' resolved `calleeIds`, batch-resolves them to callee spans +(one `s.id IN $ids` UNION-ALL over Function/Method/Constructor — keyed on the +*resolved* id, so no same-line ambiguity), seeds each callee, and runs the SAME +intra BFS within it, unioning the newly-reachable blocks into the slice. This is +**HRB context-insensitive forward closure** — the approach Joern ships (no full +SDG). Bounds: a default **3 inter-procedural function hops** (`maxDepth` caps the +per-hop intra step budget), a total node cap, and a shared `visited` set that +guarantees termination over recursion/cycles. The cross-function reach **deepens +`affectedStatements`** (the statement-level slice); the owning-symbol `byDepth` +stays a single collapsed bucket (block-hops are not call-hops). A pre-namespace-v4 +index (no `calleeIds` column) yields no callee ids, so the descent is a no-op and +the result degrades cleanly to the prior intra-only behavior. + +**Soundness caveats** (also stamped verbatim into the result `note` whenever the +slice crosses a hop): + +1. **Context-insensitive.** A dependence may be attributed to a callee only + reachable from a *different* call site of the same function (bounded + over-inclusion — the same imprecision the call-graph mode already has). +2. **Return-value ascent IS captured (CALL_SUMMARY); out-param / exception + ascent deferred.** A caller statement that depends on a callee's RETURN value + is now in the slice when the callee carries a persisted `CALL_SUMMARY` + return-flow summary (FU-C): the descent re-seeds the caller's continuation from + the call block, and FU-B-2 surfaces the dependent call/continuation statements + at statement granularity (the self-edge def→use walk). What remains deferred: + out-parameter / mutated-argument ascent, callee-written shared / captured + variables, and exception ascent (a throw the callee raises that the caller + catches) — these need an alias / try-catch model. A pre-FU-C (v3) `--pdg` index + has no `CALL_SUMMARY` edges, so return-value ascent is absent there until a + re-index (the result `note` steers to it). +3. **No cross-boundary alias model.** Aliasing of arguments/heap across the call + boundary is not modeled. +4. **Precision is bounded by the call RESOLVER's precision.** Multi-candidate + dispatch and C++ overload under-resolution flow through faithfully — the + descent is sound (it never drops a real target), but it inherits exactly the + resolver's precision, no more and no less. + +### U2 — dynamic mutation oracle (independent ground-truth cross-check) + +The PR's **#1 declared validity threat is annotation circularity** (the manual +`intra_AIS` risks being reconciled against the very traversal it scores). U2 adds +an **independent, CI-runnable check**: a real **dynamic forward slice** computed +by **value-diff**, not by reading the static slice. It lives in +`mutation-oracle.mjs` (substrate) + the pure scorers in `metrics.mjs`, gated +behind a new `--mutation` flag. **It is bench-additive: no `src/` change, no +schema change; the default report run + its F1 `--check` gate + fingerprints are +byte-identical without `--mutation`.** + +**Why value-diff, not coverage.** Per Voas's PIE model (TSE'92), an observable +fault needs **P**ropagation + **I**nfection + **E**xecution. Coverage is only E; +*dependence* requires I+P = an actual VALUE CHANGE at a downstream point. So the +oracle's `behavioral_AIS` is the set of statements whose **observed value +changed** when the criterion line was mutated — a genuine +[Agrawal-Horgan PLDI'90](https://doi.org/10.1145/93542.93576) **dynamic slice**, +not a coverage trace. The static⊇dynamic soundness relation (Tip'95) then says a +sound static slicer must contain the dynamic slice on the executed paths, so +`B ⊆ slice` is the recall expectation. + +**Per fixture, the oracle:** + +1. **Mutates the criterion line only** (≤ 4 mutants, line-scoped regex operators: + AOR `+ - * / %`, ROR `> < <= >= === !==`, LCR `|| && !`, CRP numeric-literal → + `k+1`/`0`, UOI negate-RHS when no operator is flippable). EQUIVALENT mutants + (empty `behavioral_AIS`) and syntactically-invalid mutants are discarded. +2. **Derives inputs** with a tiny TYPE-DRIVEN generator from the criterion fn's + params — `number → [5,-3,0]`, `number[] → [[1,2,3],[-1,-2],[]]`, `boolean → + [true,false]`, `string → ['a','b','z']` (multi-input covers both branch arms). + The tuples used are recorded in the sidecar. +3. **Instruments the ORIGINAL TS AST** with Babel (`@babel/parser` + `traverse` + + `generator`, `retainLines:true` so loc lines stay 1-based `filePath:line` — the + SAME space as the slice, no source-map needed): a value-transparent + `__trace(EXPR, line, filePath, occ)` wraps VariableDeclarator init / + AssignmentExpression RHS / ReturnStatement arg / CallExpression and returns + `EXPR` unchanged. Runs original + each mutant via **tsx dynamic-import** on the + SAME inputs from the SAME temp working copy the analyze step used. +4. `behavioral_AIS = { filePath:line where serialize(orig) != serialize(mut) }` + for some input/occurrence (value changed / appeared / disappeared), **EXCLUDING + the criterion line**, unioned over inputs then over non-equivalent mutants. A + deterministic serializer handles `undefined`/`NaN`/`±Infinity`/stable object key + order. + +**The metric (`--mutation`, report-only this landing).** Let `slice = +pdgLineCis(results.pdg.affectedStatements)` (the SAME live static slice the F1 +metric scores), `B = behavioral_AIS`, `M = intra_AIS`: + +- **`mutation_recall = |B ∩ slice| / |B|`** (pure `mutationRecall` in + `metrics.mjs`). `recall < 1.0` ⇒ `B ∖ slice` is a statement the oracle PROVED + depends on the criterion that the static slice MISSED. `B ∖ slice` is printed + explicitly and **every** such line is classified as **(a) known-U1-no-ascent-gap**, + **(b) driver/model artifact** (block-coalescing interior of a coalesced + BasicBlock, or the upstream-fixture oracle-direction mismatch), or **(c) novel + recall hole** — so a reader is never misled (see *Caveats* below). +- **Circularity cross-check: `B ∖ M`** (pure `circularityDiff`). Non-empty on an + **intra** fixture ⇒ the manual annotation missed a real dependence — the headline + independent evidence U2 exists to produce. Reported as a **WARN with the lines**; + it does **not** fail. (On inter/mixed fixtures `intra_AIS` is empty-by-design, so + `B ∖ M` there is expected cross-function reach, labelled as such, not a miss.) +- **Precision is NOT gated.** `slice ∖ B` is expected sound over-approximation + (the static slice legitimately over-includes); `|slice ∖ B|` is reported + informationally only. + +**Phasing — report-then-gate.** `measure.mjs --mutation --check` prints a +`Gate 4 (mutation recall, REPORT-ONLY)` line + the numbers but **does NOT +`process.exit(1)`** on `recall < 1.0` this landing. Flipping it to a hard gate is a +one-flag change: `--mutation-strict` already fails the build on **NOVEL** holes +only (block-coalescing artifacts and documented U1 gaps are excluded by the same +classifier the report uses). + +**Caveats handled.** + +- **R1 — the two `upstream` fixtures** (`reassignSum`, `filterPositive`) are a + forward-oracle mismatch. The oracle runs in its native downstream sense and does + the circularity cross-check, but the recall **gate is NOT applied** to them; this + is **printed** as `oracle-direction-excluded`, never a silent skip. +- **R2 — U1's DOCUMENTED no-ascent gap.** A caller statement depending on a callee + RETURN/out-param/thrown-exception is not in the intra slice without + `CALL_SUMMARY`. A `recall < 1.0` from callee-effect ascent into the caller + continuation is a **KNOWN gap**, not a novel bug — the report classifies it + `known-U1-no-ascent-gap` with the lines. `nobody-interface-excluded` (no body) + stays oracle-excluded; `intra-overloaded-callee` runs as **id-discrimination + corroboration** (mutating the alpha arm changes `Alpha.process`'s output, not + `Beta.process`'s), included as corroboration — **not** an AIS recall case. + +**Artifacts.** All instrumented mutants are generated under an `os.tmpdir()` dir +(`gn-impact-pdg-mut-*`), **never inside `fixtures/`**. The only persisted new file +per fixture is `mutation-ground-truth.json` (a regenerated audit cache; +`provenance:'mutation'`; **separate from the manual `ground-truth.json`, never +overwriting it**). It is data, matched by no vitest project (the default glob is +`test/**/*.test.ts`); the integration test carries a **tripwire** asserting no +`*.test.ts` ever appears under `bench/impact-pdg/`. + +**Runtime.** Each fixture costs one extra `analyze --pdg` child process (the same +substrate the F1 loop uses) plus a handful of in-process tsx dynamic-imports +(original + ≤ 4 mutants × the input tuples), all on tiny fixtures — roughly +**4–7 s/fixture**, so the full `--mutation` pass adds on the order of a minute over +the base run. The oracle runs **once** (not median-of-K): the value-diff is the +load-bearing signal, deterministic, not substrate-noise-prone like F1. + +**Cross-language sweep is a separate task.** The fixtures are TypeScript (the +maturest CFG/PDG support here) and the instrumenter is TS-AST-based. Extending the +oracle to the other languages requires re-indexing per-language fixture corpora +under `--pdg` and a per-language instrumenter — a separate re-indexing task, out of +scope for this landing. + +**Reproduce:** + +```sh +node --import tsx bench/impact-pdg/measure.mjs --mutation # per-fixture recall + circularity rows +node --import tsx bench/impact-pdg/measure.mjs --mutation --check # + report-only Gate 4 +node --import tsx bench/impact-pdg/measure.mjs --mutation --check --mutation-strict # hard-fail on NOVEL holes +``` + +### Re-baseline (after a reviewed accuracy or ground-truth change) + +1. `node --import tsx bench/impact-pdg/measure.mjs --json` and read + `annotationFingerprint` + `strata[scope][mode].f1`. +2. Copy those into `baselines.json` (`annotationFingerprint`, the `f1Bands` + cells), bump `analyzerVersion` if the analyzer moved, adjust `epsilon` only + deliberately. +3. Confirm `--check` is green. + +The fixtures are also validated by the integration test +`test/integration/impact-pdg-fixtures.test.ts` (schema well-formedness + a smoke +test that each fixture analyzes under `--pdg` and the criterion function produces +its declared CDG / REACHING_DEF edges — a zero-edge criterion has unmeasurable +ground truth). diff --git a/gitnexus/bench/impact-pdg/baselines.json b/gitnexus/bench/impact-pdg/baselines.json new file mode 100644 index 000000000..d6259c98d --- /dev/null +++ b/gitnexus/bench/impact-pdg/baselines.json @@ -0,0 +1,21 @@ +{ + "_doc": "U7 impact-PDG accuracy baselines. THREE NON-byte-identity gates (KTD10 + U9): (1) annotationFingerprint — an order-independent digest over the curated ground-truth set (INCLUDING criterion.line, the PDG slice seed, AND the U9 `idBridge` block); any unreviewed edit to a ground-truth.json trips it (re-baseline deliberately after review). (2) f1Bands — a ONE-SIDED F1 regression band per mode per scope: a DROP below (band - epsilon) fails; improvements pass freely. A `null` band means F1 is genuinely undefined for that (scope,mode) on this corpus — the gate skips it (nothing to regress against). (3) U9 resolved-id soundness axis — every fixture carrying an `idBridge` ground-truth block must prove EXACTLY its `idBridge.idProven` set statement-precise (via the resolved symbol id) while the leaf-NAME match strictly over-attributes the eliminated collision id (`idBridge.fpEliminated`). The expected sets live in the ground-truth `idBridge` block, covered by Gate 1's fingerprint, so Gate 3 has no separate baseline number. measure.mjs --check applies median-of-K across GN_IMPACT_PDG_K runs before comparing, so substrate flakiness (the real-analyze lane, F5) cannot trip the band. Re-baseline: `node --import tsx bench/impact-pdg/measure.mjs --json` → copy annotationFingerprint and strata[scope][mode].f1 here, bump analyzerVersion if the analyzer moved.", + "analyzerVersion": "1.6.7", + "epsilon": 0.05, + "annotationFingerprint": "f5792b91c0e2da2b80dbac1cbbee0219aa532806234646c35fa3edfc17d4e700", + "f1Bands": { + "intra": { + "callgraph": null, + "pdg": 1.0 + }, + "inter": { + "callgraph": 1.0, + "pdg": null + }, + "mixed": { + "callgraph": 1.0, + "pdg": 1.0 + } + }, + "_f1BandsNote": "FU-B-2 re-reconciled: the intra slice is now STATEMENT-granular (the persisted REACHING_DEF edge carries the FULL ordered list of def/use source-line pairs for its (block-pair, binding) group, and the projection walks the self-edge def->use chain forward to fixpoint to recover a coalesced block's interior statements — including a SAME-BINDING reassignment chain `acc@24->acc@25->acc@26`, which all share one deduped edge and need the whole pair list, not just the first pair). The annotationFingerprint MOVED deliberately (d5cb0045->f5792b91): three intra fixtures had their intra_AIS re-reconciled UP to the now-measurable statement set — intra-dataflow-chain {10}->{8,9,10}, intra-control-guard {9,11,13}->{9,11,12,13}, intra-dataflow-reassign {6,7}->{6,7,8} — each restoring the original source-derived belief the prior block-granularity reconciliation had under-counted. Per-fixture justification (NOT a metric re-fit): (a) intra-dataflow-chain (downstream) is independently corroborated — the U2 dynamic value-diff oracle had ALREADY proved behavioral_AIS={8,9,10} (and flagged 8/9 as a manual-annotation circularity miss) BEFORE the edit; (b) intra-control-guard (downstream) is justified by source semantics — line 12 `const z = y + 1` is control-dependent on the guard AND data-dependent on `y@11` in the reached post-guard body block — and likewise corroborated by the downstream U2 oracle; (c) intra-dataflow-reassign is UPSTREAM, so U2 is oracle-direction-EXCLUDED and does NOT justify it — line 8 `total = total + b` is kept purely on SOURCE SEMANTICS: it is the immediately-reaching definition of the criterion use `return total`@9 (the last unkilled def of `total` reaching line 9), a genuine upstream reaching-def. After re-reconciliation intra/pdg stays F1=1.0 (P=1.0 R=1.0 FPIS=0 across all 7 intra fixtures) and mixed/pdg is F1=1.0 (the FU-A intra-tag keeps cross-function callee lines on the inter axis; the recovered interior intra statements match exactly). U7-rework numbers (statement-anchored PDG slice). The two engines are scored at DIFFERENT granularities against DIFFERENT ground truths: PDG at intra-procedural LINE granularity vs intra_AIS (seeded on criterion.line), call-graph at inter-procedural SYMBOL granularity vs inter_AIS. NON-NULL bands: intra/pdg=1.0 (the line-seeded slice exactly reproduces intra_AIS across all 6 intra fixtures — FPIS=0, FNIS=0), mixed/pdg=1.0 (see below), inter/callgraph=1.0 (full cross-function callee recall), mixed/callgraph=1.0 (reaches every callee). mixed/pdg is back at 1.0 after FU-A: the U1 inter-procedural slice still UNIONS the cross-function (callee) reach into affectedStatements, but each statement now carries a projection-only scope:'intra'|'inter' tag (intra iff its owning function file + 1-based start line match the criterion's), and the intra-line axis is scored against the intra-tagged statements only (pdgLineCis(...,'intra')). So the callee lines are no longer counted as intra-axis FPIS (precision climbs 0.468->1.0) while recall stays 1.0 (FNIS=0 — every intra_AIS line is still in the slice). The cross-function reach is preserved on the separate inter symbol axis; the separated per-axis view is in the report's 'Unified impact axes' section. The one-sided band still catches any FURTHER drop (a real recall regression, or added intra over-approximation). NULL cells: intra/callgraph (a self-contained function calls no other symbol → CIS and inter_AIS both empty → F1 n/a) and inter/pdg (a pure-inter router has an empty intra_AIS by design; the line-seeded slice returns the router's own control-dependent statements as FPIS, precision 0, recall n/a → F1 n/a). The intra_AIS of 6 fixtures was reconciled against the live traversal during the rework (CFG block-coalescing of straight-line statements + the combined CDG+RD reverse slice picking up data deps); see each ground-truth.json rationale for the per-fixture correction. These four non-null bands are the live regression guard." +} diff --git a/gitnexus/bench/impact-pdg/blast-radius.mjs b/gitnexus/bench/impact-pdg/blast-radius.mjs new file mode 100644 index 000000000..45cc3994f --- /dev/null +++ b/gitnexus/bench/impact-pdg/blast-radius.mjs @@ -0,0 +1,361 @@ +/** + * Real-code blast-radius / localization probe — the "is PDG-mode impact actually + * better than callgraph-only?" evidence harness. + * + * `real-code.mjs` checks that unified `mode:'pdg'` PRESERVES callgraph symbol + * reach and how much it costs. This script answers the sharper question: when you + * change a single statement inside a real function, how much does PDG NARROW the + * impact set versus the pre-PDG answer ("you changed something in F → inspect all + * of F")? It samples real functions from an already-indexed repo and, per + * function, compares: + * + * - intra axis (the localization win): |PDG statement slice| vs |whole function + * body| (block units). A ratio < 1 means PDG points at a subset of the body + * instead of the whole thing. Correctness of that subset is NOT proven here — + * it is anchored by the AIS-backed `measure.mjs` gate, which shows the + * line-seeded slice is exact (intra/mixed PDG F1 = 1.0, FPIS = FNIS = 0). So + * a smaller slice is a genuine over-approximation cut, not a dropped-truth + * risk. + * - inter axis (the honest non-win): the PDG interprocedural symbol set vs the + * callgraph symbol set for the same target. They are equal by design (PDG + * bridges interprocedural reach through the call graph), so this probe + * surfaces any divergence rather than assuming it. + * - cost: callgraph vs PDG latency. + * + * This is a quality proxy on real code (no curated AIS), exactly like + * `real-code.mjs`. Read magnitudes as directional; the correctness claim lives in + * `measure.mjs`. + * + * Methodology note: the seed anchor is an EARLY-interior block (index + * floor(M/3)). For a downstream/forward slice that is a conservative, + * slice-maximizing choice — it understates rather than inflates the localization + * win — so the measured cut is a lower bound on a typical interior edit. + */ +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { fileURLToPath } from 'node:url'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const REPO_ROOT = path.resolve(__dirname, '..', '..'); + +function readOption(argv, name, fallback = undefined) { + const eq = argv.find((arg) => arg.startsWith(`--${name}=`)); + if (eq) return eq.slice(name.length + 3); + const idx = argv.indexOf(`--${name}`); + if (idx >= 0 && idx + 1 < argv.length) return argv[idx + 1]; + return fallback; +} + +function hasFlag(argv, name) { + return argv.includes(`--${name}`); +} + +export function median(xs) { + if (xs.length === 0) return null; + const sorted = [...xs].sort((a, b) => a - b); + const mid = Math.floor(sorted.length / 2); + return sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2; +} + +function round(value, digits = 3) { + if (value === null || value === undefined || Number.isNaN(value)) return null; + const scale = 10 ** digits; + return Math.round(value * scale) / scale; +} + +function fmt(value, digits = 2) { + return value === null || value === undefined ? 'n/a' : Number(value).toFixed(digits); +} + +/** + * Parse a GitNexus `cypher` markdown table into row objects. Only used on columns + * that cannot contain a `|` (ids, identifiers, integers) so the split is safe. + */ +export function parseMarkdownRows(markdown) { + if (!markdown) return []; + const lines = markdown.split('\n').filter((l) => l.trim().startsWith('|')); + if (lines.length < 2) return []; + const head = lines[0] + .split('|') + .slice(1, -1) + .map((s) => s.trim()); + return lines.slice(2).map((l) => { + const cells = l + .split('|') + .slice(1, -1) + .map((s) => s.trim()); + const o = {}; + head.forEach((h, i) => (o[h] = cells[i])); + return o; + }); +} + +/** Stable symbol-id set from an impact `byDepth` record (mirrors real-code.mjs). */ +export function symbolSetFromByDepth(byDepth) { + const out = new Set(); + for (const items of Object.values(byDepth ?? {})) { + for (const item of items ?? []) { + if (!item || typeof item !== 'object') continue; + if (typeof item.id === 'string' && item.id.length > 0) { + out.add(item.id); + continue; + } + const name = typeof item.name === 'string' ? item.name : '(unknown)'; + const filePath = typeof item.filePath === 'string' ? item.filePath : '(unknown)'; + out.add(`${name}@${filePath}`); + } + } + return out; +} + +/** + * Pure aggregation over the per-function measurements. Kept dependency-free so the + * deterministic unit test can assert the arithmetic without analyze/DB. + */ +export function summarizeBlastRadius(cases) { + const ratios = cases.map((c) => c.ratio).filter((v) => v !== null && v !== undefined); + return { + n: cases.length, + localization: { + medianSliceOverBody: round(median(ratios)), + meanSliceOverBody: ratios.length + ? round(ratios.reduce((a, b) => a + b, 0) / ratios.length) + : null, + medianBodyBlocks: median(cases.map((c) => c.bodyBlocks)), + medianSliceBlocks: median(cases.map((c) => c.sliceBlocks)), + casesSliceSmallerThanBody: cases.filter((c) => c.sliceBlocks < c.bodyBlocks).length, + }, + interSymbol: { + casesPdgFindsMore: cases.filter((c) => c.pdgOnly > 0).length, + casesPdgFindsFewer: cases.filter((c) => c.cgOnly > 0).length, + casesIdentical: cases.filter((c) => c.pdgOnly === 0 && c.cgOnly === 0).length, + totalPdgOnlySymbols: cases.reduce((a, c) => a + c.pdgOnly, 0), + totalCgOnlySymbols: cases.reduce((a, c) => a + c.cgOnly, 0), + }, + // Statement-precise inter-procedural reach (the axis-3 precision win): the + // proven subset invoked from the changed line's slice, vs the full callgraph + // reach. Only the with-slice cases can discriminate; empty-slice/upstream + // cases preserve full reach (precision 1) and are reported separately. + statementPrecise: { + casesWithSlice: cases.filter((c) => c.sliceBlocks > 0).length, + casesTighterThanCallgraph: cases.filter((c) => c.statementPreciseSymbols < c.callgraphSymbols) + .length, + medianStatementPrecision: round( + median(cases.map((c) => c.statementPrecision).filter((v) => v !== null && v !== undefined)), + ), + medianPreciseSymbols: median(cases.map((c) => c.statementPreciseSymbols)), + medianCallgraphSymbols: median(cases.map((c) => c.callgraphSymbols)), + }, + latency: { + medianCallgraphMs: round(median(cases.map((c) => c.callgraphMs))), + medianPdgMs: round(median(cases.map((c) => c.pdgMs))), + medianPdgOverCallgraph: round( + median( + cases + .map((c) => (c.callgraphMs > 0 ? c.pdgMs / c.callgraphMs : null)) + .filter((v) => v !== null), + ), + ), + }, + }; +} + +async function cypherRows(backend, repo, query) { + const res = await backend.callTool('cypher', { repo, query }); + return parseMarkdownRows(res?.markdown); +} + +async function run() { + const argv = process.argv.slice(2); + const repo = readOption(argv, 'repo', 'GitNexus'); + const sample = Math.max(1, Number(readOption(argv, 'sample', '120'))); + const minBlocks = Math.max(2, Number(readOption(argv, 'min-blocks', '6'))); + const src = readOption(argv, 'src', 'gitnexus/src/'); + const direction = readOption(argv, 'direction', 'downstream'); + const depth = Math.max(1, Number(readOption(argv, 'depth', '3'))); + const limit = Math.max(1, Number(readOption(argv, 'limit', '200'))); + const json = hasFlag(argv, 'json'); + + const { LocalBackend } = await import( + path.join(REPO_ROOT, 'src', 'mcp', 'local', 'local-backend.ts') + ); + const backend = new LocalBackend(); + const initialized = await backend.init(); + if (!initialized) + throw new Error('no indexed repositories found; run gitnexus analyze --pdg first'); + + try { + // Candidate functions + methods with a body worth localizing. + const candidateQuery = (label) => + `MATCH (f:${label}) WHERE f.filePath STARTS WITH '${src}' AND f.endLine > f.startLine + 18 ` + + `RETURN f.name AS name, f.filePath AS filePath, f.startLine AS startLine, ` + + `f.endLine AS endLine, '${label}' AS kind`; + let candidates = [ + ...(await cypherRows(backend, repo, candidateQuery('Function'))), + ...(await cypherRows(backend, repo, candidateQuery('Method'))), + ].filter((c) => c.name && /^[A-Za-z_$][\w$]*$/.test(c.name)); + // Stride-sample for file diversity instead of taking the first N. + const stride = Math.max(1, Math.floor(candidates.length / (sample * 3))); + candidates = candidates.filter((_, i) => i % stride === 0); + + const cases = []; + let degraded = 0; + for (const c of candidates) { + if (cases.length >= sample) break; + const lo = Number(c.startLine); + const hi = Number(c.endLine); + if (!Number.isFinite(lo) || !Number.isFinite(hi)) continue; + + // The function's OWN blocks (id prefix fnStartLine == lo+1, 1-based) — the + // line range alone would also capture nested closures. + const blockRows = await cypherRows( + backend, + repo, + `MATCH (b:BasicBlock) WHERE b.filePath = '${c.filePath}' AND b.startLine >= ${lo} ` + + `AND b.startLine <= ${hi + 1} RETURN b.id AS id, b.startLine AS startLine ORDER BY b.startLine`, + ); + const fnLine1b = String(lo + 1); + const own = blockRows.filter((r) => { + const parts = r.id.split(':'); + return parts[parts.length - 3] === fnLine1b; + }); + const bodyBlocks = own.length; + if (bodyBlocks < minBlocks) continue; + const startLines = own + .map((r) => Number(r.startLine)) + .filter(Number.isFinite) + .sort((a, b) => a - b); + const anchor = startLines[Math.max(1, Math.floor(bodyBlocks / 3))]; + if (!Number.isFinite(anchor)) continue; + + const base = { + repo, + target: c.name, + file_path: c.filePath, + kind: c.kind, + direction, + maxDepth: depth, + limit, + includeTests: true, + }; + + let t = performance.now(); + const cg = await backend.callTool('impact', { ...base, mode: 'callgraph' }); + const callgraphMs = performance.now() - t; + if (cg?.error) continue; + + t = performance.now(); + const pdg = await backend.callTool('impact', { ...base, mode: 'pdg', line: anchor }); + const pdgMs = performance.now() - t; + if (pdg?.error) continue; + if (pdg?.pdgLayer && pdg.pdgLayer !== 'ready') { + degraded++; + continue; + } + if (pdg?.epistemic === 'pdg-no-block-at-line') continue; + + const sliceBlocks = pdg?.affectedStatementCount ?? 0; + const cgSyms = symbolSetFromByDepth(cg?.byDepth ?? {}); + const pdgSyms = symbolSetFromByDepth( + pdg?.interproceduralByDepth ?? pdg?.pdgInterprocedural?.byDepth ?? {}, + ); + // Statement-precise (proven) inter-procedural reach: the subset invoked + // from the criterion's dependence slice. Tighter than callgraph when the + // changed line reaches only some of the function's callees. + const preciseSyms = symbolSetFromByDepth( + pdg?.pdgInterprocedural?.statementPreciseByDepth ?? {}, + ); + const pdgOnly = [...pdgSyms].filter((x) => !cgSyms.has(x)).length; + const cgOnly = [...cgSyms].filter((x) => !pdgSyms.has(x)).length; + + cases.push({ + name: c.name, + kind: c.kind, + file: c.filePath, + anchor, + bodyBlocks, + sliceBlocks, + ratio: bodyBlocks ? round(sliceBlocks / bodyBlocks) : null, + callgraphSymbols: cgSyms.size, + pdgSymbols: pdgSyms.size, + statementPreciseSymbols: preciseSyms.size, + statementPrecision: + typeof pdg?.pdgInterprocedural?.statementPrecision === 'number' + ? round(pdg.pdgInterprocedural.statementPrecision) + : null, + pdgOnly, + cgOnly, + epistemic: pdg?.epistemic ?? null, + callgraphMs: round(callgraphMs, 1), + pdgMs: round(pdgMs, 1), + }); + } + + const summary = summarizeBlastRadius(cases); + const report = { + repo, + direction, + sample: cases.length, + minBlocks, + degradedSkipped: degraded, + generatedAt: new Date().toISOString(), + note: 'Real-code localization proxy: slice-vs-body magnitude only. Correctness is anchored by measure.mjs (AIS-backed).', + summary, + cases, + }; + + if (json) { + process.stdout.write(JSON.stringify(report, null, 2) + '\n'); + return; + } + + const loc = summary.localization; + const inter = summary.interSymbol; + const prec = summary.statementPrecise; + const lat = summary.latency; + const lines = []; + lines.push('=== impact-PDG real-code blast-radius / localization probe ==='); + lines.push( + `repo ${repo} | direction ${direction} | functions ${cases.length} | minBlocks ${minBlocks}`, + ); + lines.push(''); + lines.push( + `Localization (axis: tighter): PDG slice is a median ${fmt(loc.medianSliceOverBody)} of the ` + + `whole function body (mean ${fmt(loc.meanSliceOverBody)}); median body ${loc.medianBodyBlocks} ` + + `blocks -> slice ${loc.medianSliceBlocks}; ${loc.casesSliceSmallerThanBody}/${cases.length} ` + + `functions localized below whole-body.`, + ); + lines.push( + `Inter-symbol reach (axis: more callers/callees): full reach identical to callgraph on ` + + `${inter.casesIdentical}/${cases.length} functions ` + + `(pdg-only ${inter.totalPdgOnlySymbols}, callgraph-only ${inter.totalCgOnlySymbols}).`, + ); + lines.push( + `Statement-precise reach (axis: tighter cross-function): ${prec.casesTighterThanCallgraph}/` + + `${prec.casesWithSlice} with-slice functions narrow below full callgraph reach; median ` + + `statement-precision ${fmt(prec.medianStatementPrecision)} (proven median ` + + `${prec.medianPreciseSymbols} vs callgraph ${prec.medianCallgraphSymbols} symbols).`, + ); + lines.push( + `Latency (axis: faster): callgraph median ${fmt(lat.medianCallgraphMs, 1)}ms, ` + + `pdg median ${fmt(lat.medianPdgMs, 1)}ms, pdg/cg ${fmt(lat.medianPdgOverCallgraph)}x.`, + ); + lines.push(''); + lines.push( + 'Interpretation: PDG narrows the intra-procedural impact set (the slice is a ' + + 'fraction of the body); its correctness — that the narrowed set drops no real ' + + 'dependency — is the AIS-backed measure.mjs result (intra/mixed PDG F1 = 1.0). PDG ' + + 'does NOT widen cross-function reach (equal to callgraph by design) and is NOT faster.', + ); + process.stdout.write(lines.join('\n') + '\n'); + } finally { + await backend.dispose().catch(() => {}); + } +} + +if (path.resolve(process.argv[1] ?? '') === fileURLToPath(import.meta.url)) { + run().catch((err) => { + process.stderr.write(`[impact-pdg-blast-radius] ERROR: ${err?.stack || err}\n`); + process.exit(1); + }); +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-dispatcher-thin/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/inter-dispatcher-thin/ground-truth.json new file mode 100644 index 000000000..4a8184fed --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/inter-dispatcher-thin/ground-truth.json @@ -0,0 +1,33 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "dispatch", + "filePath": "src/dispatcher.ts", + "direction": "downstream", + "line": 23, + "marker": "kind === 'b'", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "inter", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [], + "inter_AIS": [ + { + "symbol": "handleA", + "filePath": "src/dispatcher.ts", + "note": "routed to when kind === 'a'; its behavior is part of dispatch's true blast radius" + }, + { + "symbol": "handleB", + "filePath": "src/dispatcher.ts", + "note": "routed to when kind === 'b'" + }, + { + "symbol": "handleDefault", + "filePath": "src/dispatcher.ts", + "note": "the fallthrough route" + } + ], + "rationale": "criterion.line=23 — the dispatcher's own first body statement `if (kind === 'a')` (source semantics: the router's work begins here). intra_AIS is EMPTY BY DESIGN: `dispatch` is a thin router whose TRUE impact is ENTIRELY cross-function — changing it affects the three handlers it routes to (handleA/handleB/handleDefault), captured in inter_AIS. The only intra statements are the routing branch returns, which carry no work of their own, so there is no meaningful intra-procedural ground truth. Step-0 reconciliation: the statement-anchored PDG slice from line 23 DOES return the routing return statements (the branch predicates control-depend on each other and each return), but those are NOT the meaningful blast radius — PDG is intra-procedural, so on a pure-inter fixture its intra slice is noise relative to the (empty) intra ground truth. This is the symmetric counterpart of call-graph's empty intra slice on the intra fixtures: each engine is blind to the other's locus. The call-graph mode (inter-procedural) finds all three handlers exactly. Anchors the `inter` stratum. NOTE: the smoke test requires `dispatch` to produce SOME CDG/REACHING_DEF edges (its routing branches do) — a zero-edge criterion would be unmeasurable; but those edges are not the AIS." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-dispatcher-thin/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/inter-dispatcher-thin/mutation-ground-truth.json new file mode 100644 index 000000000..cb0b80d01 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/inter-dispatcher-thin/mutation-ground-truth.json @@ -0,0 +1,43 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "dispatch", + "filePath": "src/dispatcher.ts", + "line": 23 + }, + "paramTypes": ["string", "number"], + "inputs": [ + ["a", 5], + ["a", -3], + ["a", 0], + ["b", 5], + ["b", -3], + ["b", 0] + ], + "behavioral_AIS": [ + "src/dispatcher.ts:10", + "src/dispatcher.ts:14", + "src/dispatcher.ts:18", + "src/dispatcher.ts:24", + "src/dispatcher.ts:27", + "src/dispatcher.ts:29" + ], + "mutants": [ + { + "op": "ROR", + "text": " if (kind !== 'a') {", + "equivalent": false, + "diffLines": [ + "src/dispatcher.ts:10", + "src/dispatcher.ts:14", + "src/dispatcher.ts:18", + "src/dispatcher.ts:24", + "src/dispatcher.ts:27", + "src/dispatcher.ts:29" + ] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-dispatcher-thin/src/dispatcher.ts b/gitnexus/bench/impact-pdg/fixtures/inter-dispatcher-thin/src/dispatcher.ts new file mode 100644 index 000000000..1b7af560b --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/inter-dispatcher-thin/src/dispatcher.ts @@ -0,0 +1,30 @@ +// Thin cross-function dispatcher. `dispatch`'s TRUE impact is entirely +// cross-function: it just routes to `handleA` / `handleB` / `handleDefault`. +// Intra-procedural PDG over `dispatch` returns ~no truly-affected statements +// of interest (the routing branch returns are control-dependent on the +// selector, but the *meaningful* blast radius — the work — lives in the +// callees), so PDG inter-AIS recall here is ~0 BY DESIGN. The call-graph mode +// is the right tool: its inter-procedural reach finds the three handlers. + +export function handleA(payload: number): number { + return payload + 1; +} + +export function handleB(payload: number): number { + return payload * 2; +} + +export function handleDefault(payload: number): number { + return payload; +} + +export function dispatch(kind: string, payload: number): number { + // criterion: a thin router. Changing it affects the callees it routes to. + if (kind === 'a') { + return handleA(payload); + } + if (kind === 'b') { + return handleB(payload); + } + return handleDefault(payload); +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-facade-delegate/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/inter-facade-delegate/ground-truth.json new file mode 100644 index 000000000..f36f8387e --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/inter-facade-delegate/ground-truth.json @@ -0,0 +1,33 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "processOrder", + "filePath": "src/facade.ts", + "direction": "downstream", + "line": 21, + "marker": "enrich(order)", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "inter", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [], + "inter_AIS": [ + { + "symbol": "validate", + "filePath": "src/facade.ts", + "note": "the guard delegate; processOrder's behavior depends on it" + }, + { + "symbol": "enrich", + "filePath": "src/facade.ts", + "note": "transforms the order before persistence" + }, + { + "symbol": "persist", + "filePath": "src/facade.ts", + "note": "the terminal delegate in the sequence" + } + ], + "rationale": "criterion.line=21 — the facade's first body statement, the guard `if (!validate(order))` (source semantics: the facade's work begins here). intra_AIS is EMPTY BY DESIGN: `processOrder` sequences three delegates (validate, enrich, persist) with one guard; its true blast radius is the delegation chain — inter_AIS — not its own body, which only routes values between calls. The guard return and the local `enriched` binding carry no independent work beyond shuttling delegate results, so there is no meaningful intra ground truth. Step-0 reconciliation: the statement-anchored PDG slice from line 21 returns the guard's control-dependent body statements, but those are noise relative to the (empty) intra ground truth — PDG is intra-procedural and cannot reach the cross-function delegates. The call-graph mode walks validate->enrich->persist exactly. Anchors the `inter` stratum with a sequential-delegation shape (distinct from the dispatcher's branching shape). The single guard gives `processOrder` a CDG edge so the smoke test's edge-presence requirement is met." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-facade-delegate/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/inter-facade-delegate/mutation-ground-truth.json new file mode 100644 index 000000000..5d6fc7260 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/inter-facade-delegate/mutation-ground-truth.json @@ -0,0 +1,34 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "processOrder", + "filePath": "src/facade.ts", + "line": 21 + }, + "paramTypes": ["number"], + "inputs": [[5], [-3], [0]], + "behavioral_AIS": [ + "src/facade.ts:12", + "src/facade.ts:16", + "src/facade.ts:22", + "src/facade.ts:24", + "src/facade.ts:25" + ], + "mutants": [ + { + "op": "LCR", + "text": " if (validate(order)) {", + "equivalent": false, + "diffLines": [ + "src/facade.ts:12", + "src/facade.ts:16", + "src/facade.ts:22", + "src/facade.ts:24", + "src/facade.ts:25" + ] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-facade-delegate/src/facade.ts b/gitnexus/bench/impact-pdg/fixtures/inter-facade-delegate/src/facade.ts new file mode 100644 index 000000000..a4c93b487 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/inter-facade-delegate/src/facade.ts @@ -0,0 +1,26 @@ +// Facade that delegates to a layered set of helpers. `processOrder`'s true +// impact is cross-function: it sequences `validate`, `enrich`, and `persist`. +// Its own body is a thin sequence of calls with a single guard; the real work +// (and the real blast radius) is in the delegates. PDG intra-AIS is ~empty by +// design; the call-graph mode walks the delegation chain. + +export function validate(order: number): boolean { + return order > 0; +} + +export function enrich(order: number): number { + return order + 100; +} + +export function persist(order: number): number { + return order; +} + +export function processOrder(order: number): number { + // criterion: a facade. Its impact flows into the three delegates. + if (!validate(order)) { + return -1; + } + const enriched = enrich(order); + return persist(enriched); +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/ground-truth.json new file mode 100644 index 000000000..d15db3ef6 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/ground-truth.json @@ -0,0 +1,33 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "runPipeline", + "filePath": "src/pipeline.ts", + "direction": "downstream", + "line": 20, + "marker": "stageTransform(acc)", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "inter", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [], + "inter_AIS": [ + { + "symbol": "stageParse", + "filePath": "src/pipeline.ts", + "note": "first pipeline stage runPipeline invokes; its output feeds the next" + }, + { + "symbol": "stageTransform", + "filePath": "src/pipeline.ts", + "note": "second stage; depends on stageParse's output threaded through `acc`" + }, + { + "symbol": "stageEmit", + "filePath": "src/pipeline.ts", + "note": "terminal stage" + } + ], + "rationale": "criterion.line=20 — the driver's first body statement `let acc = seed` (source semantics: the pipeline's threading begins here). Direction is DOWNSTREAM (dependencies): changing `runPipeline` affects the callees it invokes. In GitNexus `impact` vocabulary, downstream = dependencies/callees; the stages are what the driver CALLS, so the correct tag is downstream (an earlier draft mislabeled this `upstream`, conflating the English 'upstream sources' with GitNexus's caller-direction — corrected after the harness surfaced a direction-vs-AIS contradiction). intra_AIS is EMPTY BY DESIGN: the driver threads `acc` through three stage calls (stageParse->stageTransform->stageEmit); the dependencies that matter cross function boundaries — the three stage functions — so inter_AIS holds them. The `acc` reassignments only carry delegate results, no independent computation, so there is no meaningful intra ground truth. Step-0 reconciliation: the statement-anchored PDG slice from line 20 returns the intra `acc` def->use / guard statements, but those are noise relative to the (empty) intra ground truth — PDG is intra-procedural and cannot reach the cross-function stages. The call-graph mode (downstream) reaches the three stages exactly. The `enabled` guard (added so the driver carries a CDG edge) keeps the criterion measurable for the smoke test without changing the cross-function locus. Distinct from the dispatcher (branch-routed) and facade (guarded-sequence) shapes — this is a straight pipeline driver." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/mutation-ground-truth.json new file mode 100644 index 000000000..8acebf384 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/mutation-ground-truth.json @@ -0,0 +1,47 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "runPipeline", + "filePath": "src/pipeline.ts", + "line": 20 + }, + "paramTypes": ["number", "boolean"], + "inputs": [ + [5, true], + [5, false], + [-3, true], + [-3, false], + [0, true], + [0, false] + ], + "behavioral_AIS": [ + "src/pipeline.ts:11", + "src/pipeline.ts:15", + "src/pipeline.ts:22", + "src/pipeline.ts:24", + "src/pipeline.ts:25", + "src/pipeline.ts:26", + "src/pipeline.ts:27", + "src/pipeline.ts:7" + ], + "mutants": [ + { + "op": "UOI", + "text": " let acc = -(seed);", + "equivalent": false, + "diffLines": [ + "src/pipeline.ts:11", + "src/pipeline.ts:15", + "src/pipeline.ts:22", + "src/pipeline.ts:24", + "src/pipeline.ts:25", + "src/pipeline.ts:26", + "src/pipeline.ts:27", + "src/pipeline.ts:7" + ] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/src/pipeline.ts b/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/src/pipeline.ts new file mode 100644 index 000000000..06401c62e --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/src/pipeline.ts @@ -0,0 +1,28 @@ +// A pipeline driver whose impact is cross-function: `runPipeline` calls each +// stage in turn. The stages themselves carry the work; the driver loops over +// them. Annotated UPSTREAM from the driver: "what does runPipeline depend on?" +// -> the stage functions it invokes. Intra-PDG is ~empty by design. + +export function stageParse(n: number): number { + return n + 1; +} + +export function stageTransform(n: number): number { + return n * 3; +} + +export function stageEmit(n: number): number { + return n - 2; +} + +export function runPipeline(seed: number, enabled: boolean): number { + // criterion (upstream): a driver. It depends on the three stage functions. + let acc = seed; + if (!enabled) { + return acc; // a guard so the driver itself carries a CDG edge + } + acc = stageParse(acc); + acc = stageTransform(acc); + acc = stageEmit(acc); + return acc; +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-branch/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-control-branch/ground-truth.json new file mode 100644 index 000000000..47cdc2bdf --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-branch/ground-truth.json @@ -0,0 +1,42 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "classify", + "filePath": "src/branch.ts", + "direction": "downstream", + "line": 7, + "marker": "'zero'", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "intra", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "classify", + "filePath": "src/branch.ts", + "line": 9, + "note": "`return 'pos'` is control-dependent on the `x > 0` arm" + }, + { + "symbol": "classify", + "filePath": "src/branch.ts", + "line": 10, + "note": "the `else if (x < 0)` predicate is itself control-dependent on the outer `x > 0` branch (it runs only on the false arm)" + }, + { + "symbol": "classify", + "filePath": "src/branch.ts", + "line": 11, + "note": "`return 'neg'` is control-dependent on the `x < 0` else-if arm" + }, + { + "symbol": "classify", + "filePath": "src/branch.ts", + "line": 13, + "note": "`return 'zero'` is control-dependent on the fallthrough arm" + } + ], + "inter_AIS": [], + "rationale": "criterion.line=7 — the if predicate `if (x > 0)`, the branch whose change propagates (source semantics: the predicate controls which arm runs). DOWNSTREAM CDG controller->dependent edges reach the arm returns (lines 9, 11, 13) AND the nested `else if (x < 0)` test (line 10). CORRECTION (Step-0 reconciliation): the source-derived belief was intra_AIS={9,11,13} (the three returns only), but the `else if` predicate on line 10 is ITSELF control-dependent on the outer branch — it executes only on the `x > 0` false arm — so line 10 is a genuine CDG dependent the original annotation omitted. The CFG models it as its own predicate BasicBlock (`(x < 0)`), and the slice from line 7 returns {9, 10, 11, 13} exactly. The rationale's own note already flagged line 10 as nested control flow; this corrects it from out-of-set to in-set. All dependents intra-procedural; inter_AIS empty. Exercises CDG-forward over a multi-arm branch." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-branch/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-control-branch/mutation-ground-truth.json new file mode 100644 index 000000000..0759a8ab4 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-branch/mutation-ground-truth.json @@ -0,0 +1,28 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "classify", + "filePath": "src/branch.ts", + "line": 7 + }, + "paramTypes": ["number"], + "inputs": [[5], [-3], [0]], + "behavioral_AIS": ["src/branch.ts:11", "src/branch.ts:13", "src/branch.ts:9"], + "mutants": [ + { + "op": "ROR", + "text": " if (x <= 0) {", + "equivalent": false, + "diffLines": ["src/branch.ts:11", "src/branch.ts:13", "src/branch.ts:9"] + }, + { + "op": "CRP", + "text": " if (x > 1) {", + "equivalent": true, + "diffLines": [] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-branch/src/branch.ts b/gitnexus/bench/impact-pdg/fixtures/intra-control-branch/src/branch.ts new file mode 100644 index 000000000..964064724 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-branch/src/branch.ts @@ -0,0 +1,14 @@ +// Pure intra-procedural CONTROL-dependence fixture: an if/else-if/else. +// The branch predicate controls which return statement runs. Changing the +// branch (the criterion) control-affects all three arms via CDG, all within +// the same function. + +export function classify(x: number): string { + if (x > 0) { + // criterion: the branch predicate controls the arms below + return 'pos'; // control-dependent on the branch (first arm) + } else if (x < 0) { + return 'neg'; // control-dependent on the branch (second arm) + } + return 'zero'; // control-dependent on the branch (fallthrough arm) +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-gate/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-control-gate/ground-truth.json new file mode 100644 index 000000000..3842fa4d8 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-gate/ground-truth.json @@ -0,0 +1,30 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "gate", + "filePath": "src/gate.ts", + "direction": "downstream", + "line": 10, + "marker": "!ok", + "pdgEdgeKinds": ["CDG"] + }, + "locus": "intra", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "gate", + "filePath": "src/gate.ts", + "line": 11, + "note": "`return 0` is control-dependent on the guard (true arm)" + }, + { + "symbol": "gate", + "filePath": "src/gate.ts", + "line": 13, + "note": "`return 1` is control-dependent on the guard (executes only when the guard is false)" + } + ], + "inter_AIS": [], + "rationale": "SOURCE-ONLY fixture (KTD9 annotation-circularity anchor): the intra_AIS below was written PURELY from language semantics and is deliberately NOT reconciled against the live traversal — there is no `CORRECTION (Step-0 reconciliation)` block. criterion.line=10 is the guard predicate `if (!ok)`; downstream CDG controller->dependent edges reach the true-arm `return 0` (line 11) and the fall-through `return 1` (line 13). Each arm is a single statement on its own block, so unlike the coalescing-prone sibling control fixtures there is no consecutive-statement merge to surprise the annotation: the source-derived set {11, 13} is also the measurable set. If the traversal scores this F1=1.0 it is an INDEPENDENT confirmation (the analyzer matched an un-reconciled, semantics-only annotation), not self-consistency — the one corpus data point that breaks the circularity threat the other fixtures document. All dependents are WITHIN `gate`; inter_AIS empty." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-gate/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-control-gate/mutation-ground-truth.json new file mode 100644 index 000000000..2c465674f --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-gate/mutation-ground-truth.json @@ -0,0 +1,22 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "gate", + "filePath": "src/gate.ts", + "line": 10 + }, + "paramTypes": ["boolean"], + "inputs": [[true], [false]], + "behavioral_AIS": ["src/gate.ts:11", "src/gate.ts:13"], + "mutants": [ + { + "op": "LCR", + "text": " if (ok) {", + "equivalent": false, + "diffLines": ["src/gate.ts:11", "src/gate.ts:13"] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-gate/src/gate.ts b/gitnexus/bench/impact-pdg/fixtures/intra-control-gate/src/gate.ts new file mode 100644 index 000000000..96da57e30 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-gate/src/gate.ts @@ -0,0 +1,14 @@ +// Pure intra-procedural CONTROL-dependence fixture — SOURCE-ONLY (KTD9 anchor). +// Unlike the sibling control fixtures, this fixture's intra_AIS is NOT reconciled +// against the live traversal: each guarded arm is a SINGLE statement on its own +// block, so there is no consecutive-statement coalescing to surprise the +// source-derived annotation. It exists to give the corpus one independent data +// point — if the traversal matches an annotation written purely from language +// semantics, F1=1.0 is a genuine confirmation, not self-consistency. + +export function gate(ok: boolean): number { + if (!ok) { + return 0; + } + return 1; +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-guard/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-control-guard/ground-truth.json new file mode 100644 index 000000000..e0da1984f --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-guard/ground-truth.json @@ -0,0 +1,42 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "guarded", + "filePath": "src/guard.ts", + "direction": "downstream", + "line": 7, + "marker": "y + 1", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "intra", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "guarded", + "filePath": "src/guard.ts", + "line": 9, + "note": "`return -1` is control-dependent on the guard's true arm" + }, + { + "symbol": "guarded", + "filePath": "src/guard.ts", + "line": 11, + "note": "the post-guard body block `const y = x * 2; const z = y + 1;` (lines 11-12, coalesced) runs only when the guard is false: control-dependent (block-start representative)" + }, + { + "symbol": "guarded", + "filePath": "src/guard.ts", + "line": 12, + "note": "`const z = y + 1` is control-dependent on the guard AND data-dependent on `y` (interior of the coalesced 11-12 body block; recovered statement-granular by FU-B-2 walking the `y@11->z@12` self-edge in the reached body block)" + }, + { + "symbol": "guarded", + "filePath": "src/guard.ts", + "line": 13, + "note": "`return z` is control-dependent on the guard" + } + ], + "inter_AIS": [], + "rationale": "criterion.line=7 — the guard predicate `if (!ok)` (source semantics: the guard controls whether the post-guard body runs). DOWNSTREAM CDG controller->dependent edges reach the true-arm `return -1` (line 9), the post-guard body, and `return z` (line 13). CORRECTION (FU-B-2 re-reconciliation): the CFG COALESCES the two consecutive post-guard defs `const y = x * 2` and `const z = y + 1` into ONE BasicBlock starting at line 11. BEFORE FU-B-2 the block-granular slice could not pinpoint the interior line 12, so the measurable set was {9, 11, 13}. FU-B-2 makes the slice STATEMENT-granular: the reached body block (11-12) is itself walked over its self REACHING_DEF def->use lines (`y@11->z@12`, decoded from the FU-B-2 `reason` annotation), surfacing line 12. So the measurable control/data-dependent set is now {9, 11, 12, 13} — the slice from line 7 returns exactly that, restoring the ORIGINAL source-derived belief (the block-coalescing under-count is repaired, not a metric re-fit). All dependents WITHIN `guarded`; inter_AIS empty. Canonical guard-clause CDG shape (#559); isolates the CDG-forward arm of KTD4 combined with the FU-B-2 intra-block data-dep recovery on a control-reached body block." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-guard/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-control-guard/mutation-ground-truth.json new file mode 100644 index 000000000..f7e664c7a --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-guard/mutation-ground-truth.json @@ -0,0 +1,29 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "guarded", + "filePath": "src/guard.ts", + "line": 7 + }, + "paramTypes": ["boolean", "number"], + "inputs": [ + [true, 5], + [true, -3], + [true, 0], + [false, 5], + [false, -3], + [false, 0] + ], + "behavioral_AIS": ["src/guard.ts:11", "src/guard.ts:12", "src/guard.ts:13", "src/guard.ts:9"], + "mutants": [ + { + "op": "LCR", + "text": " if (ok) {", + "equivalent": false, + "diffLines": ["src/guard.ts:11", "src/guard.ts:12", "src/guard.ts:13", "src/guard.ts:9"] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-guard/src/guard.ts b/gitnexus/bench/impact-pdg/fixtures/intra-control-guard/src/guard.ts new file mode 100644 index 000000000..b46214a23 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-guard/src/guard.ts @@ -0,0 +1,14 @@ +// Pure intra-procedural CONTROL-dependence fixture: the guard clause. +// The early-return guard predicate controls whether the post-guard body runs. +// Changing the guard (the criterion) control-affects the body statements via +// CDG controller->dependent edges, all within the same function. + +export function guarded(ok: boolean, x: number): number { + if (!ok) { + // criterion: the guard predicate controls the arms below + return -1; // control-dependent on the guard (true arm) + } + const y = x * 2; // control-dependent on the guard (false arm reaches here) + const z = y + 1; // control-dependent on the guard, data-dependent on `y` + return z; // control-dependent on the guard +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-loop/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-control-loop/ground-truth.json new file mode 100644 index 000000000..59f298bb5 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-loop/ground-truth.json @@ -0,0 +1,42 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "filterPositive", + "filePath": "src/loop.ts", + "direction": "upstream", + "line": 11, + "marker": "count + 1", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "intra", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "filterPositive", + "filePath": "src/loop.ts", + "line": 6, + "note": "the function-signature/parameter block defines `xs`, which the `for` loop iterates — an upstream reaching dependency of the increment" + }, + { + "symbol": "filterPositive", + "filePath": "src/loop.ts", + "line": 7, + "note": "`let count = 0` is the initial def of `count` reaching the increment via REACHING_DEF — an upstream data dependency" + }, + { + "symbol": "filterPositive", + "filePath": "src/loop.ts", + "line": 8, + "note": "the enclosing `for` loop guard controls whether the if and its body run (transitive CDG controller)" + }, + { + "symbol": "filterPositive", + "filePath": "src/loop.ts", + "line": 10, + "note": "the inner `if (x > 0)` predicate directly controls the `count` increment (immediate CDG controller)" + } + ], + "inter_AIS": [], + "rationale": "criterion.line=11 — the `count = count + 1` increment (source semantics: UPSTREAM asks which definitions/predicates govern this statement). CORRECTION (Step-0 reconciliation): the source-derived belief was intra_AIS={8,10} (the two CONTROL predicates only). But the statement-anchored PDG slice is a COMBINED CDG+REACHING_DEF reverse slice, so it correctly also reaches the DATA dependencies of the increment: line 7 (`let count = 0`, the initial reaching def of `count`) and line 6 (the parameter block defining `xs`, which feeds the `for` loop). The original annotation under-counted by considering only control dependence. The full upstream slice is {6, 7, 8, 10}; the slice from line 11 returns exactly that. Both controllers (8, 10) and both data deps (6, 7) are intra-procedural; inter_AIS empty. Exercises the CDG+RD-reverse slice over nested control structure." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-loop/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-control-loop/mutation-ground-truth.json new file mode 100644 index 000000000..22c297a8b --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-loop/mutation-ground-truth.json @@ -0,0 +1,28 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "filterPositive", + "filePath": "src/loop.ts", + "line": 11 + }, + "paramTypes": ["number[]"], + "inputs": [[[1, 2, 3]], [[-1, -2]], [[]]], + "behavioral_AIS": ["src/loop.ts:14"], + "mutants": [ + { + "op": "AOR", + "text": " count = count - 1; // criterion (upstream): controlled by the loop AND the if", + "equivalent": false, + "diffLines": ["src/loop.ts:14"] + }, + { + "op": "CRP", + "text": " count = count + 2; // criterion (upstream): controlled by the loop AND the if", + "equivalent": false, + "diffLines": ["src/loop.ts:14"] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-control-loop/src/loop.ts b/gitnexus/bench/impact-pdg/fixtures/intra-control-loop/src/loop.ts new file mode 100644 index 000000000..506e0a607 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-control-loop/src/loop.ts @@ -0,0 +1,15 @@ +// Pure intra-procedural CONTROL-dependence fixture, annotated UPSTREAM. +// The criterion is the loop body; upstream asks "what controls whether this +// runs?" The answer is the enclosing loop guard. Exercises the CDG-reverse arm +// of KTD4 over a loop's control structure, all within one function. + +export function filterPositive(xs: number[]): number { + let count = 0; + for (const x of xs) { + // the loop guard controls the body below + if (x > 0) { + count = count + 1; // criterion (upstream): controlled by the loop AND the if + } + } + return count; +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-accumulator/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-accumulator/ground-truth.json new file mode 100644 index 000000000..fca2ca19f --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-accumulator/ground-truth.json @@ -0,0 +1,30 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "total", + "filePath": "src/accumulator.ts", + "direction": "downstream", + "line": 8, + "marker": "sum + x", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "intra", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "total", + "filePath": "src/accumulator.ts", + "line": 10, + "note": "for-loop body re-defines/uses `sum`: data-dependent on the initial def" + }, + { + "symbol": "total", + "filePath": "src/accumulator.ts", + "line": 12, + "note": "return reads the accumulated `sum`: data-dependent on every prior def" + } + ], + "inter_AIS": [], + "rationale": "criterion.line=8 — the `sum` definition `let sum = 0`, the statement whose change propagates (chosen from source semantics: `sum` is the loop-carried accumulator). DOWNSTREAM from line 8, the REACHING_DEF def->use edges reach the in-loop redefinition/use (line 10) and the final return's use (line 12). These two lines are the only truly-affected statements and they are all WITHIN `total` — so intra_AIS is exactly {line 10, line 12}; inter_AIS is empty (the function calls nothing). Step-0 reconciliation: the statement-anchored PDG slice from line 8 returns exactly {10, 12} — an EXACT match to this source-derived annotation (no correction needed). The call-graph mode only knows `total` exists and has no inbound edges, so it reports the empty cross-function set; PDG resolves the precise dependent statements the call-graph cannot express below function granularity." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-accumulator/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-accumulator/mutation-ground-truth.json new file mode 100644 index 000000000..3885e761a --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-accumulator/mutation-ground-truth.json @@ -0,0 +1,22 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "total", + "filePath": "src/accumulator.ts", + "line": 8 + }, + "paramTypes": ["number[]"], + "inputs": [[[1, 2, 3]], [[-1, -2]], [[]]], + "behavioral_AIS": ["src/accumulator.ts:10", "src/accumulator.ts:12"], + "mutants": [ + { + "op": "CRP", + "text": " let sum = 1; // criterion: the def of `sum`", + "equivalent": false, + "diffLines": ["src/accumulator.ts:10", "src/accumulator.ts:12"] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-accumulator/src/accumulator.ts b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-accumulator/src/accumulator.ts new file mode 100644 index 000000000..6fa0705c2 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-accumulator/src/accumulator.ts @@ -0,0 +1,13 @@ +// Pure intra-procedural data-flow fixture. +// `sum` is a loop-carried accumulator: its definition flows forward to the +// next iteration's use and to the final `return`. Changing the `sum` def +// (the criterion) affects only statements within this same function via +// REACHING_DEF def->use edges. Nothing crosses a function boundary. + +export function total(xs: number[]): number { + let sum = 0; // criterion: the def of `sum` + for (const x of xs) { + sum = sum + x; // use of `sum` (and a redefinition) — data-dependent on the def + } + return sum; // use of `sum` — data-dependent +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-chain/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-chain/ground-truth.json new file mode 100644 index 000000000..a4bd78037 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-chain/ground-truth.json @@ -0,0 +1,36 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "chainCompute", + "filePath": "src/chain.ts", + "direction": "downstream", + "line": 7, + "marker": "b - 3", + "pdgEdgeKinds": ["REACHING_DEF"] + }, + "locus": "intra", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "chainCompute", + "filePath": "src/chain.ts", + "line": 8, + "note": "`const b = a * 2` is data-dependent on `a` (interior of the coalesced 7-9 block; recovered statement-granular by FU-B-2)" + }, + { + "symbol": "chainCompute", + "filePath": "src/chain.ts", + "line": 9, + "note": "`const c = b - 3` is data-dependent on `b` (transitively on `a`; interior of the coalesced 7-9 block; recovered statement-granular by FU-B-2)" + }, + { + "symbol": "chainCompute", + "filePath": "src/chain.ts", + "line": 10, + "note": "`return c` is transitively data-dependent on `a` via the chain — a distinct downstream BasicBlock" + } + ], + "inter_AIS": [], + "rationale": "criterion.line=7 — the def of `a` (`const a = input + 1`), the changed statement (source semantics: `a` seeds the straight-line def->use chain a->b->c->return). DOWNSTREAM from line 7. CORRECTION (FU-B-2 re-reconciliation): the CFG COALESCES consecutive straight-line statements into ONE BasicBlock: lines 7-9 (`const a`, `const b`, `const c`) share a single block whose start line is 7 — the criterion's own seed block. BEFORE FU-B-2 the block-granular slice could not pinpoint the interior statements 8/9 (they lived inside the seed block, which the traversal excludes), so the measurable intra_AIS was {10} only. FU-B-2 makes the intra slice STATEMENT-granular: the persisted REACHING_DEF edge now carries its def/use SOURCE LINES (codec annotation on `reason`), and the projection walks the self-edge def->use chain (a@7->b@8->c@9) forward from the criterion line to recover the interior dependents. So the measurable intra_AIS is now {8, 9, 10} — the slice returns exactly {8,9,10}, an EXACT match to the corrected annotation. This is an independently-justified ground-truth correction, NOT a metric re-fit: the U2 dynamic value-diff oracle PROVED behavioral_AIS = {8,9,10} (mutation-ground-truth.json) and flagged lines 8/9 as a manual-annotation circularity miss BEFORE this edit — the dynamic oracle and the static slice agree. Isolates the RD-forward arm of the KTD4 truth table; inter_AIS empty (no calls)." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-chain/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-chain/mutation-ground-truth.json new file mode 100644 index 000000000..f69c62f2e --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-chain/mutation-ground-truth.json @@ -0,0 +1,28 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "chainCompute", + "filePath": "src/chain.ts", + "line": 7 + }, + "paramTypes": ["number"], + "inputs": [[5], [-3], [0]], + "behavioral_AIS": ["src/chain.ts:10", "src/chain.ts:8", "src/chain.ts:9"], + "mutants": [ + { + "op": "AOR", + "text": " const a = input - 1; // criterion: the def of `a`", + "equivalent": false, + "diffLines": ["src/chain.ts:10", "src/chain.ts:8", "src/chain.ts:9"] + }, + { + "op": "CRP", + "text": " const a = input + 2; // criterion: the def of `a`", + "equivalent": false, + "diffLines": ["src/chain.ts:10", "src/chain.ts:8", "src/chain.ts:9"] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-chain/src/chain.ts b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-chain/src/chain.ts new file mode 100644 index 000000000..053bc1d14 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-chain/src/chain.ts @@ -0,0 +1,11 @@ +// Pure intra-procedural data-flow fixture: a straight-line def->use chain. +// Changing the first definition `a` flows through `b` and `c` to the return, +// all within one function. There are no branches and no calls — the purest +// REACHING_DEF chain. + +export function chainCompute(input: number): number { + const a = input + 1; // criterion: the def of `a` + const b = a * 2; // data-dependent on `a` + const c = b - 3; // data-dependent on `b` (transitively on `a`) + return c; // data-dependent on `c` +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-reassign/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-reassign/ground-truth.json new file mode 100644 index 000000000..b07599cc7 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-reassign/ground-truth.json @@ -0,0 +1,36 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "reassignSum", + "filePath": "src/reassign.ts", + "direction": "upstream", + "line": 9, + "marker": "total + b", + "pdgEdgeKinds": ["REACHING_DEF"] + }, + "locus": "intra", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "reassignSum", + "filePath": "src/reassign.ts", + "line": 6, + "note": "the function-signature/parameter block defines `a`, whose value flows into `total = a`; it is a genuine reaching def upstream of the criterion use" + }, + { + "symbol": "reassignSum", + "filePath": "src/reassign.ts", + "line": 7, + "note": "the coalesced def block `let total = a; total = total + b;` (lines 7-8) — block-start representative; `let total = a` is a reaching def of the criterion use `return total`" + }, + { + "symbol": "reassignSum", + "filePath": "src/reassign.ts", + "line": 8, + "note": "`total = total + b` is the immediately-reaching def of the criterion use (interior of the coalesced 7-8 def block; recovered statement-granular by FU-B-2 walking the `total@7->total@8` self-edge in the reached def block)" + } + ], + "inter_AIS": [], + "rationale": "criterion.line=9 — the `return total` use (source semantics: UPSTREAM asks which definitions reach this use of `total`). CORRECTION (FU-B-2 re-reconciliation): the CFG COALESCES the two consecutive defs `let total = a` and `total = total + b` into ONE BasicBlock starting at line 7, and REACHING_DEF also reaches the parameter block (line 6, defining `a`). BEFORE FU-B-2 the block-granular slice could not pinpoint the interior line 8, so the measurable set was {6, 7}. FU-B-2 makes the slice STATEMENT-granular: the reached def block (7-8) is walked over its self REACHING_DEF def->use lines (`total@7->total@8`, decoded from the FU-B-2 `reason` annotation), surfacing line 8. So the measurable upstream slice is now {6 (param def of `a`), 7 (coalesced block start), 8 (the interior reassign)} — the slice from line 9 returns exactly that, restoring the ORIGINAL source-derived belief {7,8} plus the param def 6 (the block-coalescing under-count is repaired, not a metric re-fit). The intra-block forward def->use walk is direction-agnostic by design; here line 8 IS upstream-relevant (it is the immediately-reaching def of the criterion use). All intra-procedural; inter_AIS empty. Isolates the RD-reverse arm of the KTD4 truth table over a reassigned variable, with FU-B-2 interior recovery on the reached coalesced def block." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-reassign/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-reassign/mutation-ground-truth.json new file mode 100644 index 000000000..db15e941b --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-reassign/mutation-ground-truth.json @@ -0,0 +1,29 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "reassignSum", + "filePath": "src/reassign.ts", + "line": 9 + }, + "paramTypes": ["number", "number"], + "inputs": [ + [5, 5], + [5, -3], + [5, 0], + [-3, 5], + [-3, -3], + [-3, 0] + ], + "behavioral_AIS": [], + "mutants": [ + { + "op": "UOI", + "text": " return -(total); // criterion: use of `total`", + "equivalent": true, + "diffLines": [] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-reassign/src/reassign.ts b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-reassign/src/reassign.ts new file mode 100644 index 000000000..8f3a4314d --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-dataflow-reassign/src/reassign.ts @@ -0,0 +1,10 @@ +// Pure intra-procedural data-flow fixture, annotated UPSTREAM. +// The criterion is the final `return total` (a use of `total`); upstream asks +// "what does this use depend on?" The answer is every reaching definition of +// `total` within the function. Exercises the reverse RD arm of KTD4. + +export function reassignSum(a: number, b: number): number { + let total = a; // first def of `total` — reaches the use below + total = total + b; // second def of `total` (uses prior) — reaches the use below + return total; // criterion: use of `total` +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-overloaded-callee/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-overloaded-callee/ground-truth.json new file mode 100644 index 000000000..18b582ae2 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-overloaded-callee/ground-truth.json @@ -0,0 +1,24 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "route", + "filePath": "src/route.ts", + "direction": "downstream", + "line": 32, + "marker": "alpha.process(x)", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "inter", + "pdgScoring": "exclude", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [], + "inter_AIS": [], + "idBridge": { + "seedLine": 32, + "idProven": ["Method:src/route.ts:Alpha.process#1"], + "nameWouldProve": ["Method:src/route.ts:Alpha.process#1", "Method:src/route.ts:Beta.process#1"], + "fpEliminated": ["Method:src/route.ts:Beta.process#1"] + }, + "rationale": "RESOLVED-SYMBOL-ID SOUNDNESS gate (plan 2026-06-18-001 U9; Covers R1, R5). Two methods share the LEAF name `process` but resolve to DISTINCT ids — `Method:src/route.ts:Alpha.process#1` and `Method:src/route.ts:Beta.process#1`. `route` invokes BOTH (one in each `if` arm), so the symbol-graph BFS reaches both as depth-1 direct callees. criterion.line=32 is `out = alpha.process(x)` — its block calls ONLY `alpha.process`, so the dependence slice's `BasicBlock.calleeIds` carries `Alpha.process#1` alone (verified: `calleeIds = 'Method:src/route.ts:Alpha.process#1'`). The RESOLVED-ID bridge therefore proves EXACTLY `Alpha.process#1` (`pdgInterprocedural.statementPreciseByDepth[1]` = that single id; `Beta.process#1` is reached but `unproven-bridge`). The leaf-NAME bridge over the SAME reached items + the SAME slice would prove BOTH (the slice `callees` = {process}; both callees share that leaf), over-attributing `Beta.process#1` — the collision false-positive (`fpEliminated`). `idBridge` records the gate: id-proven == the single correct id; name-match proves 2. `pdgScoring:\"exclude\"` keeps this fixture OUT of the intra/inter strata F1 bands (its locus is genuinely cross-function and its intra_AIS/inter_AIS are not the measured quantity — the id-vs-name soundness diff is); the dedicated id-bridge axis in measure.mjs scores it and `--check` gates the statement-precise id set against `idBridge.idProven`. inter_AIS is intentionally empty: this fixture exists to gate the id-vs-name discrimination, not the cross-function recall the sibling inter fixtures already cover. The ids embed the repo-relative `src/route.ts` path (stable: the harness analyzes a temp copy whose repo-relative paths match)." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-overloaded-callee/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/intra-overloaded-callee/mutation-ground-truth.json new file mode 100644 index 000000000..c9588da07 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-overloaded-callee/mutation-ground-truth.json @@ -0,0 +1,29 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "route", + "filePath": "src/route.ts", + "line": 32 + }, + "paramTypes": ["number", "boolean"], + "inputs": [ + [5, true], + [5, false], + [-3, true], + [-3, false], + [0, true], + [0, false] + ], + "behavioral_AIS": ["src/route.ts:38"], + "mutants": [ + { + "op": "UOI", + "text": " out = -(alpha.process(x));", + "equivalent": false, + "diffLines": ["src/route.ts:38"] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/intra-overloaded-callee/src/route.ts b/gitnexus/bench/impact-pdg/fixtures/intra-overloaded-callee/src/route.ts new file mode 100644 index 000000000..bbd8a5228 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/intra-overloaded-callee/src/route.ts @@ -0,0 +1,39 @@ +// Resolved-symbol-id SOUNDNESS fixture (U9 — plan 2026-06-18-001). +// +// Two callees share the LEAF name `process` but resolve to DISTINCT symbol ids +// (`Alpha.process` vs `Beta.process`). `route` invokes BOTH — so the symbol-graph +// BFS reaches both as direct callees — but each call lives in its OWN control +// block (separate `if` arms), so the SEED line's dependence slice contains only +// the block that calls `alpha.process`. The id bridge proves exactly +// `Alpha.process` for that seed line; the leaf-NAME bridge would prove BOTH +// (`process` is in the slice block's `callees`), over-attributing `Beta.process`. +// This fixture gates that the resolved-id bridge eliminates the same-leaf-name +// collision false-positive. + +export class Alpha { + process(value: number): number { + return value + 1; + } +} + +export class Beta { + process(value: number): number { + return value * 2; + } +} + +export function route(x: number, useAlpha: boolean): number { + const alpha = new Alpha(); + const beta = new Beta(); + let out = 0; + if (useAlpha) { + // SEED line: this block calls ONLY alpha.process — its slice carries the + // resolved id `Alpha.process`, NOT `Beta.process`. + out = alpha.process(x); + } else { + // Independent block: beta.process is reached by the BFS (shared `process` + // leaf name) but is NOT on the alpha-seed line's dependence slice. + out = beta.process(x); + } + return out; +} diff --git a/gitnexus/bench/impact-pdg/fixtures/mixed-compute-and-emit/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/mixed-compute-and-emit/ground-truth.json new file mode 100644 index 000000000..87d3d38fe --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/mixed-compute-and-emit/ground-truth.json @@ -0,0 +1,42 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "computeAndEmit", + "filePath": "src/mixed.ts", + "direction": "downstream", + "line": 12, + "marker": "score + v", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "mixed", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "computeAndEmit", + "filePath": "src/mixed.ts", + "line": 14, + "note": "loop body `score = score + v` re-defines/uses `score`: data-dependent" + }, + { + "symbol": "computeAndEmit", + "filePath": "src/mixed.ts", + "line": 16, + "note": "`level = score > 10 ? ...` is data-dependent on the accumulated `score`" + }, + { + "symbol": "computeAndEmit", + "filePath": "src/mixed.ts", + "line": 17, + "note": "the `emit(level)` arg is data-dependent on `level`/`score`" + } + ], + "inter_AIS": [ + { + "symbol": "emit", + "filePath": "src/mixed.ts", + "note": "called with the computed level; cross-function reach" + } + ], + "rationale": "criterion.line=12 — the `score` def `let score = 0` (source semantics: `score` seeds the loop-carried data chain). Criterion `computeAndEmit` is mixed-locus. DOWNSTREAM from line 12, the `score` def drives the loop-carried data chain (line 14), feeds the threshold `level` (line 16), which feeds the `emit` call arg (line 17): all intra-procedural data dependence, so intra_AIS holds those three statements. Separately, the call to `emit` is a cross-function reach captured in inter_AIS. The two sets are disjoint (lines within `computeAndEmit` vs the distinct symbol `emit`). Step-0 reconciliation: the statement-anchored PDG slice from line 12 returns {14, 16, 17} — an EXACT match to this source-derived intra_AIS (no correction needed); call-graph reaches {emit} exactly. Distinct from mixed-validate-then-call: this case's intra dependence is data-flow-dominated (accumulator + ternary) rather than guard-dominated, broadening mixed-stratum coverage." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/mixed-compute-and-emit/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/mixed-compute-and-emit/mutation-ground-truth.json new file mode 100644 index 000000000..73dbd6309 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/mixed-compute-and-emit/mutation-ground-truth.json @@ -0,0 +1,22 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "computeAndEmit", + "filePath": "src/mixed.ts", + "line": 12 + }, + "paramTypes": ["number[]"], + "inputs": [[[1, 2, 3]], [[-1, -2]], [[]]], + "behavioral_AIS": ["src/mixed.ts:14"], + "mutants": [ + { + "op": "CRP", + "text": " let score = 1; // def; loop-carried accumulation below", + "equivalent": false, + "diffLines": ["src/mixed.ts:14"] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/mixed-compute-and-emit/src/mixed.ts b/gitnexus/bench/impact-pdg/fixtures/mixed-compute-and-emit/src/mixed.ts new file mode 100644 index 000000000..582372d49 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/mixed-compute-and-emit/src/mixed.ts @@ -0,0 +1,18 @@ +// Mixed-locus fixture: `computeAndEmit` has a real intra-procedural data-flow +// computation (a running `score` accumulated in a loop, then thresholded) AND a +// cross-function reach (it calls `emit`). The criterion's blast radius spans +// both loci. + +export function emit(level: string): string { + return '[' + level + ']'; +} + +export function computeAndEmit(values: number[]): string { + // criterion: mixed. Intra: score accumulation + threshold; inter: emit(). + let score = 0; // def; loop-carried accumulation below + for (const v of values) { + score = score + v; // data-dependent on prior `score` + } + const level = score > 10 ? 'high' : 'low'; // data-dependent on `score` + return emit(level); // cross-function reach; arg data-dependent on `level` +} diff --git a/gitnexus/bench/impact-pdg/fixtures/mixed-guarded-dispatch/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/mixed-guarded-dispatch/ground-truth.json new file mode 100644 index 000000000..019aa970b --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/mixed-guarded-dispatch/ground-truth.json @@ -0,0 +1,47 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "route", + "filePath": "src/mixed.ts", + "direction": "downstream", + "line": 15, + "marker": "key === 0", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "mixed", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "route", + "filePath": "src/mixed.ts", + "line": 16, + "note": "the guard `if (urgent || key === 0)` is data-dependent on `key` and controls the arms" + }, + { + "symbol": "route", + "filePath": "src/mixed.ts", + "line": 18, + "note": "`return fast(n)` is control-dependent on the guard's true arm" + }, + { + "symbol": "route", + "filePath": "src/mixed.ts", + "line": 20, + "note": "`return slow(n)` is control-dependent on the complementary arm" + } + ], + "inter_AIS": [ + { + "symbol": "fast", + "filePath": "src/mixed.ts", + "note": "callee on the urgent/even route; cross-function reach" + }, + { + "symbol": "slow", + "filePath": "src/mixed.ts", + "note": "callee on the fallthrough route; cross-function reach" + } + ], + "rationale": "criterion.line=15 — the `key` def `const key = n % 2` (source semantics: `key` feeds the guard predicate below). Criterion `route` is mixed-locus with BOTH a guard whose predicate reads the intra def `key` (line 15 -> line 16) and control-gates both return arms (lines 18, 20), AND two cross-function reaches (fast, slow). DOWNSTREAM from line 15: intra_AIS holds the guard (16) + the two control-dependent return statements (18, 20); inter_AIS holds the two callees. Disjoint by construction (lines within `route` vs distinct symbols). Step-0 reconciliation: the statement-anchored PDG slice from line 15 returns {16, 18, 20} — an EXACT match to this source-derived intra_AIS; call-graph reaches {fast, slow} exactly. This mixed case combines control AND data dependence intra-procedurally with branching cross-function reach — the richest mixed shape, complementing the data-dominated and guard-dominated mixed cases." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/mixed-guarded-dispatch/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/mixed-guarded-dispatch/mutation-ground-truth.json new file mode 100644 index 000000000..aa149492f --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/mixed-guarded-dispatch/mutation-ground-truth.json @@ -0,0 +1,35 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "route", + "filePath": "src/mixed.ts", + "line": 15 + }, + "paramTypes": ["number", "boolean"], + "inputs": [ + [5, true], + [5, false], + [-3, true], + [-3, false], + [0, true], + [0, false] + ], + "behavioral_AIS": ["src/mixed.ts:10", "src/mixed.ts:18", "src/mixed.ts:20", "src/mixed.ts:6"], + "mutants": [ + { + "op": "AOR", + "text": " const key = n * 2; // def; used in the guard below", + "equivalent": true, + "diffLines": [] + }, + { + "op": "CRP", + "text": " const key = n % 3; // def; used in the guard below", + "equivalent": false, + "diffLines": ["src/mixed.ts:10", "src/mixed.ts:18", "src/mixed.ts:20", "src/mixed.ts:6"] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/mixed-guarded-dispatch/src/mixed.ts b/gitnexus/bench/impact-pdg/fixtures/mixed-guarded-dispatch/src/mixed.ts new file mode 100644 index 000000000..b052753e7 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/mixed-guarded-dispatch/src/mixed.ts @@ -0,0 +1,21 @@ +// Mixed-locus fixture: `route` has an intra-procedural guard chain that both +// computes a `key` (data flow) and control-gates which callee runs (control +// flow), AND reaches two callees cross-function. Both loci are non-trivial. + +export function fast(n: number): number { + return n; +} + +export function slow(n: number): number { + return n + n; +} + +export function route(n: number, urgent: boolean): number { + // criterion: mixed. Intra: key + guard; inter: fast()/slow(). + const key = n % 2; // def; used in the guard below + if (urgent || key === 0) { + // control + data dependent on `key` + return fast(n); // control-dependent on the guard; cross-function reach + } + return slow(n); // control-dependent on the complementary arm; cross-function reach +} diff --git a/gitnexus/bench/impact-pdg/fixtures/mixed-validate-then-call/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/mixed-validate-then-call/ground-truth.json new file mode 100644 index 000000000..4c74ce213 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/mixed-validate-then-call/ground-truth.json @@ -0,0 +1,48 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "handleRequest", + "filePath": "src/mixed.ts", + "direction": "downstream", + "line": 13, + "marker": "normalized + 1", + "pdgEdgeKinds": ["REACHING_DEF", "CDG"] + }, + "locus": "mixed", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [ + { + "symbol": "handleRequest", + "filePath": "src/mixed.ts", + "line": 14, + "note": "the guard `if (normalized < 0)` is data-dependent on `normalized` and controls the arms" + }, + { + "symbol": "handleRequest", + "filePath": "src/mixed.ts", + "line": 16, + "note": "`return -1` is control-dependent on the guard" + }, + { + "symbol": "handleRequest", + "filePath": "src/mixed.ts", + "line": 18, + "note": "`const adjusted = normalized + 1` is data-dependent on `normalized`" + }, + { + "symbol": "handleRequest", + "filePath": "src/mixed.ts", + "line": 19, + "note": "the `persist(adjusted)` arg is data-dependent on `adjusted`/`normalized`; the return is control-dependent on the guard" + } + ], + "inter_AIS": [ + { + "symbol": "persist", + "filePath": "src/mixed.ts", + "note": "called with the adjusted value; cross-function reach" + } + ], + "rationale": "criterion.line=13 — the `normalized` def `const normalized = raw * 2` (source semantics: `normalized` seeds both the data chain and the guard). Criterion `handleRequest` is mixed-locus. DOWNSTREAM from line 13, the `normalized` def drives BOTH a data chain (guard test line 14, `adjusted` line 18, the call arg line 19) and a control structure (the guard controls lines 16 and 19). So intra_AIS holds the four control/data-dependent statements within the function. Separately, `handleRequest` calls `persist` — a genuine cross-function reach — so inter_AIS holds `persist`. Step-0 reconciliation: the statement-anchored PDG slice from line 13 returns {14, 16, 18, 19} — an EXACT match to this source-derived intra_AIS; call-graph reaches {persist} exactly. This is the case where the two modes are complementary: PDG resolves the intra statement set precisely, call-graph reaches the callee. intra_AIS and inter_AIS are disjoint by construction (one is lines within `handleRequest`, the other is a different symbol `persist`)." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/mixed-validate-then-call/mutation-ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/mixed-validate-then-call/mutation-ground-truth.json new file mode 100644 index 000000000..1b995cd17 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/mixed-validate-then-call/mutation-ground-truth.json @@ -0,0 +1,28 @@ +{ + "schemaVersion": 1, + "provenance": "mutation", + "criterion": { + "name": "handleRequest", + "filePath": "src/mixed.ts", + "line": 13 + }, + "paramTypes": ["number"], + "inputs": [[5], [-3], [0]], + "behavioral_AIS": ["src/mixed.ts:18", "src/mixed.ts:19", "src/mixed.ts:8"], + "mutants": [ + { + "op": "AOR", + "text": " const normalized = raw / 2; // def; data-dependent uses below", + "equivalent": false, + "diffLines": ["src/mixed.ts:18", "src/mixed.ts:19", "src/mixed.ts:8"] + }, + { + "op": "CRP", + "text": " const normalized = raw * 3; // def; data-dependent uses below", + "equivalent": false, + "diffLines": ["src/mixed.ts:18", "src/mixed.ts:19", "src/mixed.ts:8"] + } + ], + "skipped": null, + "note": "AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS (Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. Independent cross-check of the manual ground-truth.json — NOT a hand annotation." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/mixed-validate-then-call/src/mixed.ts b/gitnexus/bench/impact-pdg/fixtures/mixed-validate-then-call/src/mixed.ts new file mode 100644 index 000000000..f30d7d01b --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/mixed-validate-then-call/src/mixed.ts @@ -0,0 +1,20 @@ +// Mixed-locus fixture: `handleRequest` has BOTH genuine intra-procedural +// dependence (a normalized value computed and reused across statements, guarded +// by a validity check) AND a cross-function reach (it calls `persist`). +// Changing it affects intra statements (the normalize->guard->use chain) AND +// the callee. + +export function persist(value: number): number { + return value; +} + +export function handleRequest(raw: number): number { + // criterion: mixed. Intra: normalized flows through the guard to the call. + const normalized = raw * 2; // def; data-dependent uses below + if (normalized < 0) { + // control-dependent guard + return -1; // control-dependent on the guard + } + const adjusted = normalized + 1; // data-dependent on `normalized` + return persist(adjusted); // cross-function reach; arg data-dependent on `adjusted` +} diff --git a/gitnexus/bench/impact-pdg/fixtures/nobody-interface-excluded/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/nobody-interface-excluded/ground-truth.json new file mode 100644 index 000000000..b72313898 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/nobody-interface-excluded/ground-truth.json @@ -0,0 +1,15 @@ +{ + "schemaVersion": 1, + "criterion": { + "name": "Shape", + "filePath": "src/nobody.ts", + "direction": "downstream" + }, + "locus": "n/a", + "pdgScoring": "exclude", + "provenance": "manual", + "analyzerVersion": "1.6.7", + "intra_AIS": [], + "inter_AIS": [], + "rationale": "This is the KTD6 no-body case. `Shape` (an interface), `ShapeName` (a type alias), and `AbstractShape.perimeter` (an abstract method) have no CFG body, so they emit ZERO PDG (CDG/REACHING_DEF) edges. Their `intra_AIS` is genuinely undefined under PDG, not empty-because-safe — a confident `impactedCount:0/LOW` would be the exact false-safe the impact tool exists to prevent (#2129/#1858 lineage). The case is tagged `pdgScoring: \"exclude\"` and `locus: \"n/a\"` so the U7 harness drops it from PDG precision/recall denominators (and logs the exclusion, never scoring it 0/0). It exists to assert the exclusion path is honored — it is the only fixture the smoke test exempts from the 'criterion produces CDG+REACHING_DEF edges' requirement. intra_AIS and inter_AIS are both empty by definition (no body)." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/nobody-interface-excluded/src/nobody.ts b/gitnexus/bench/impact-pdg/fixtures/nobody-interface-excluded/src/nobody.ts new file mode 100644 index 000000000..a5cae82b2 --- /dev/null +++ b/gitnexus/bench/impact-pdg/fixtures/nobody-interface-excluded/src/nobody.ts @@ -0,0 +1,16 @@ +// No-body fixture (the KTD6 case). An interface, a type alias, and an abstract +// method have NO CFG body, so they produce ZERO PDG edges. PDG mode cannot +// score these — a bare `impactedCount:0` would read as a confident "safe", +// the exact false-safe impact exists to prevent. The harness EXCLUDES this +// case from PDG scoring (pdgScoring: "exclude"); it exists to assert the +// exclusion path, not to be measured. + +export interface Shape { + area(): number; // criterion: an interface method declaration — no body, no CFG +} + +export type ShapeName = 'circle' | 'square'; + +export abstract class AbstractShape { + abstract perimeter(): number; // abstract method — no body +} diff --git a/gitnexus/bench/impact-pdg/gate-mutation-recall.mjs b/gitnexus/bench/impact-pdg/gate-mutation-recall.mjs new file mode 100644 index 000000000..5d3e44a0d --- /dev/null +++ b/gitnexus/bench/impact-pdg/gate-mutation-recall.mjs @@ -0,0 +1,45 @@ +// CI gate for the nightly impact-PDG mutation oracle (#2227 tri-review, U11). +// +// The oracle (`measure.mjs --mutation --json`) uploads a machine report as a +// nightly artifact, but nothing read it back — a realized-recall regression +// would silently sit in an artifact nobody opens. This gate: +// 1. always writes a recall summary to the GitHub job summary (visible on the +// run without downloading the artifact), and +// 2. fails the job when the MINIMUM realized recall across scored mutation +// cases drops below MUTATION_RECALL_FLOOR (tunable env, conservative +// default) — so a mutant the slicer stops catching surfaces as a red run. +// +// Usage: node bench/impact-pdg/gate-mutation-recall.mjs [report.json] +import fs from 'node:fs'; + +const reportPath = process.argv[2] ?? 'mutation-report.json'; +const floor = Number(process.env.MUTATION_RECALL_FLOOR ?? '0.5'); + +const report = JSON.parse(fs.readFileSync(reportPath, 'utf8')); +const checks = Array.isArray(report?.mutation?.checks) ? report.mutation.checks : []; +const scored = checks.filter((c) => typeof c.recall === 'number'); +const recalls = scored.map((c) => c.recall); +const min = recalls.length ? Math.min(...recalls) : null; +const mean = recalls.length ? recalls.reduce((a, b) => a + b, 0) / recalls.length : null; +const below = scored.filter((c) => c.recall < floor); +const fmt = (x) => (x === null ? 'n/a' : x.toFixed(3)); + +const summary = [ + '## impact-PDG mutation oracle', + '', + `- scored cases: ${scored.length} of ${checks.length}`, + `- min realized recall: ${fmt(min)} (floor ${floor})`, + `- mean realized recall: ${fmt(mean)}`, + `- cases below floor: ${below.length}${below.length ? ' — ' + below.map((c) => c.name).join(', ') : ''}`, + '', +].join('\n'); + +if (process.env.GITHUB_STEP_SUMMARY) { + fs.appendFileSync(process.env.GITHUB_STEP_SUMMARY, summary + '\n'); +} +process.stdout.write(summary + '\n'); + +if (min !== null && min < floor) { + console.error(`Mutation recall regression: min realized recall ${fmt(min)} < floor ${floor}`); + process.exit(1); +} diff --git a/gitnexus/bench/impact-pdg/measure.mjs b/gitnexus/bench/impact-pdg/measure.mjs new file mode 100644 index 000000000..65d3a7675 --- /dev/null +++ b/gitnexus/bench/impact-pdg/measure.mjs @@ -0,0 +1,1390 @@ +/** + * U7 — PDG-vs-call-graph impact ACCURACY measurement harness. + * + * Runs BOTH `impact` engines over the curated U6 ground-truth fixtures and + * scores each at its NATIVE granularity: + * - `mode:'pdg'` is seeded on the criterion's STATEMENT (`line: criterion.line`) + * and scored at intra-procedural LINE granularity against `intra_AIS` + * (CIS_pdg = the `affectedStatements` line set); + * - `mode:'callgraph'` is scored at inter-procedural SYMBOL granularity against + * `inter_AIS` (CIS = the reported symbol set). + * It computes precision/recall/F1 stratified by impact locus (intra/inter/mixed) + * plus cross-mode set-diffs, prints a stratified report ending in a plain- + * language DECISION RECOMMENDATION, and (under `--check`) gates regressions with + * two NON-byte-identity gates. The two engines answer DIFFERENT questions at + * DIFFERENT granularities — the report shows both, neither strictly dominates. + * + * ── Substrate (the load-bearing mechanism — KTD9/R8; plan U7 "Substrate + * decision") ────────────────────────────────────────────────────────────── + * `runPipelineFromRepo` is in-memory and never persists; `impact` queries a + * PERSISTED `repo.lbugPath` + a `meta.pdg` stamp. There is no exported + * `runAnalyze` (the entrypoint `analyzeCommand` calls `process.exit`, unusable + * in a loop), and the test-suite `vi.mock` registry bridge is vitest-only. So: + * REAL analyze via a temp `GITNEXUS_HOME`, mock-free. Per fixture: + * 1. point `process.env.GITNEXUS_HOME` at a per-run temp dir (honored by + * `repo-manager.getGlobalDir()` — it roots the registry; the per-repo DB + * lands in `/.gitnexus/`, so fixtures are copied to a temp + * working dir to keep the source tree clean); + * 2. SHELL OUT to the real CLI as a child process: + * node --import tsx src/cli/index.ts analyze --pdg --skip-git --index-only + * (child-process isolation sidesteps `process.exit`; real `saveMeta` + + * `registerRepo` land in the temp home; workers spawn from `dist/`, so the + * harness builds `dist/` first — run `node scripts/build.js`); + * 3. `new LocalBackend(); await init()` resolves the fixture via the REAL + * registry (the parent process ALSO sets `GITNEXUS_HOME` so init reads the + * temp registry, not the user's ~/.gitnexus); + * 4. `callTool('impact', …)` ×2 — callgraph (symbol BFS) and pdg (seeded on + * `line: criterion.line` so it returns the statement-anchored slice); + * 5. teardown the temp home + copy. + * + * The `repo` arg is the absolute fixture-copy PATH (tier-1 path match in + * `resolveRepoFromCache`) — unambiguous, no name collisions. + * + * ── Granularity / CIS-AIS framing ────────────────────────────────────────── + * See `metrics.mjs`. PDG = intra-procedural LINE granularity vs `intra_AIS`; + * call-graph = inter-procedural SYMBOL granularity vs `inter_AIS`. The two + * engines measure different scopes; both are now non-empty. + * + * Build-free: `node --import tsx bench/impact-pdg/measure.mjs`. Runtime budget + * and re-baseline instructions: see README.md. + */ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { spawnSync } from 'node:child_process'; +import { fileURLToPath, pathToFileURL } from 'node:url'; + +import { + symbolKey, + pdgLineCis, + intraLineAis, + score, + aggregate, + aisByScope, + unifiedAis, + callgraphUnifiedCis, + pdgUnifiedCis, + composeUnifiedCis, + scoreUnifiedAxes, + aggregateUnifiedScores, + fingerprintAnnotationSet, + mutationRecall, + circularityDiff, + isEquivalentMutant, + fingerprintMutationSet, + median, +} from './metrics.mjs'; +// U2 dynamic-oracle substrate: regex-mutate the criterion line, Babel-instrument +// the original TS AST, run original + mutants on type-driven inputs via tsx +// dynamic-import, value-diff into a behavioral (dynamic forward slice) AIS. Gated +// behind --mutation. `deriveBehavioralAis` / `writeMutationSidecar` (and their heavy +// @babel/* deps) are LAZY dynamic-imported inside run()'s --mutation branch, NOT +// statically here: a top-level import pulls Babel into this module's graph, and a +// test that imports measure.mjs for its pure helpers (impact-pdg-id-bridge-gate.test.ts) +// then crashes the vitest worker under full-suite memory pressure (the documented +// static-heavy-import-crashes-module-load pattern). The default report never loads it. +// U9 resolved-id soundness axis: reuse the U8 PURE bridge-predicate replica +// (`bridgeProvenSets`) and the id-vs-name set diff (`scoreIdVsName`) so the gate +// computes the NAME-match counterfactual the exact same way the realized-FP +// harness does — one source of truth for "what name-match would prove". +import { bridgeProvenSets, scoreIdVsName, reachedItemKey } from './name-collision.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const REPO_ROOT = path.resolve(__dirname, '..', '..'); // gitnexus/ +const FIXTURES_DIR = path.join(__dirname, 'fixtures'); +const BASELINE_PATH = path.join(__dirname, 'baselines.json'); +const CLI_ENTRY = path.join(REPO_ROOT, 'src', 'cli', 'index.ts'); + +const SCOPES = ['intra', 'inter', 'mixed']; +const MODES = ['callgraph', 'pdg']; +const UNIFIED_MODES = ['callgraph', 'pdg', 'composed-current']; + +// ── F3 minimum-corpus floor (KTD9): below this the harness reports DIRECTION +// only, never a headline decimal verdict. Mirrors the U6 schema test's floor. +const FLOOR_PER_STRATUM = 3; +const FLOOR_TOTAL = 12; + +const sha256 = (s) => crypto.createHash('sha256').update(s).digest('hex'); + +// ── fixture loading ──────────────────────────────────────────────────────── + +function loadFixtures(filter) { + const names = fs + .readdirSync(FIXTURES_DIR, { withFileTypes: true }) + .filter((d) => d.isDirectory()) + .map((d) => d.name) + .filter((n) => !filter || filter.includes(n)) + .sort(); + return names.map((name) => { + const dir = path.join(FIXTURES_DIR, name); + const gt = JSON.parse(fs.readFileSync(path.join(dir, 'ground-truth.json'), 'utf8')); + return { name, dir, gt, excluded: gt.pdgScoring === 'exclude' }; + }); +} + +// ── substrate: analyze a fixture into a temp GITNEXUS_HOME, run both modes ─── + +/** + * Copy the fixture src into a temp working dir, analyze it with `--pdg` as a + * child process (real persistence into the temp GITNEXUS_HOME), then drive both + * impact modes through a fresh LocalBackend. Returns the raw impact results + + * the working-copy path (so the criterion file paths line up with the + * annotations, which are repo-relative `src/...`). `pdgOn` toggles `--pdg` so + * the degraded-index scenario (KTD7) can be exercised. + */ +async function analyzeAndImpact(fx, home, { pdgOn = true } = {}) { + const work = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-impact-pdg-work-')); + fs.cpSync(path.join(fx.dir, 'src'), path.join(work, 'src'), { recursive: true }); + + const env = { ...process.env, GITNEXUS_HOME: home }; + const args = ['--import', 'tsx', CLI_ENTRY, 'analyze', work, '--skip-git', '--index-only']; + if (pdgOn) args.push('--pdg'); + const an = spawnSync(process.execPath, args, { + env, + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'pipe'], + timeout: 180000, + }); + if (an.status !== 0) { + fs.rmSync(work, { recursive: true, force: true }); + throw new Error( + `analyze failed for ${fx.name} (exit ${an.status}): ${(an.stderr || an.stdout || '').slice(-600)}`, + ); + } + + // The parent process must see the temp GITNEXUS_HOME too — LocalBackend.init() + // reads the REAL registry under getGlobalDir() (no mock). A fresh backend per + // fixture avoids cross-fixture pool/registry caching. + // + // FIX 8: wrap the post-analyze body in try/finally. The analyze-failure path + // above already cleans `work`; if `backend.callTool()` (or init/import) THROWS + // here, `work` would otherwise leak. On success we return `work` so the caller + // can run its own validation + cleanup; on throw we remove it before rethrow. + let succeeded = false; + try { + process.env.GITNEXUS_HOME = home; + const { LocalBackend } = await import( + path.join(REPO_ROOT, 'src', 'mcp', 'local', 'local-backend.ts') + ); + const backend = new LocalBackend(); + await backend.init(); + + // callgraph: symbol→symbol BFS (no statement anchor). pdg: SEEDED on the + // criterion's statement line so it returns the dependence slice — the U7 + // rework's central change. A whole-symbol pdg slice (no `line`) is empty by + // design; `criterion.line` is the 1-based source line of the changed + // statement (set from source semantics, validated in Step 0). + const results = { + callgraph: await backend.callTool('impact', { + repo: work, + target: fx.gt.criterion.name, + direction: fx.gt.criterion.direction, + mode: 'callgraph', + }), + pdg: await backend.callTool('impact', { + repo: work, + target: fx.gt.criterion.name, + direction: fx.gt.criterion.direction, + mode: 'pdg', + line: fx.gt.criterion.line, + }), + }; + succeeded = true; + return { work, results }; + } finally { + // Only clean up on the throwing path — on success the caller owns `work` + // (it runs `validateFixture(fx, work, ...)` then removes it in its finally). + if (!succeeded) fs.rmSync(work, { recursive: true, force: true }); + } +} + +/** + * Flatten a CALLGRAPH impact result's byDepth into canonical SYMBOL keys (the + * CIS_callgraph, scored against `inter_AIS`). Unresolved shadow entries are kept + * (keyed by file) so a recall loss is never hidden. + */ +function callgraphCisFromResult(res) { + const items = Object.values(res?.byDepth ?? {}).flat(); + const keys = new Set(); + const meta = { unresolved: 0, ambiguous: 0 }; + for (const it of items) { + if (it?.unresolved) { + meta.unresolved += 1; + keys.add(symbolKey('(unresolved)', it.filePath)); + continue; + } + if (it?.ambiguous) meta.ambiguous += 1; + keys.add(symbolKey(it.name, it.filePath)); + } + return { keys, meta }; +} + +/** + * Extract the PDG statement-line CIS (`:` keys) from a pdg + * impact result's `affectedStatements`, plus the diagnostic fields the report + * surfaces (the slice's epistemic marker / note / block count). This is the + * U7-rework CIS: the dependent STATEMENTS the change at `criterion.line` reaches. + */ +function pdgCisFromResult(res) { + const inter = callgraphCisFromResult({ + byDepth: res?.interproceduralByDepth ?? res?.pdgInterprocedural?.byDepth ?? {}, + }); + return { + lineKeys: pdgLineCis(res?.affectedStatements), + // FU-A: the intra-line axis is scored against intra-tagged statements only, + // so U1's cross-function (inter) reach is no longer counted as intra FPIS. + // The full `lineKeys` stays for diagnostics; inter reach lives on the + // separate symbol axis (`symbolKeys` / `interproceduralByDepth`). + intraLineKeys: pdgLineCis(res?.affectedStatements, 'intra'), + symbolKeys: inter.keys, + meta: { + affectedStatementCount: res?.affectedStatementCount ?? 0, + blockCount: res?.blockCount ?? null, + criterionLine: res?.criterionLine ?? null, + epistemic: res?.epistemic ?? null, + note: res?.note ?? null, + interprocedural: res?.pdgInterprocedural ?? null, + }, + }; +} + +/** + * Callee NAME set + resolved-ID set for an EXACT dependence slice (seed ∪ + * reachable blocks), read off the persisted `BasicBlock.callees` / `.calleeIds` + * via the raw pool `exec` — exactly the two sets the bridge unions over the slice + * in `local-backend.ts`. Mirrors `sliceCalleeSetsOf` in name-collision.mjs but + * runs through `exec` (the harness already holds the pool open for Step 0), so + * the name counterfactual is computed from the SAME persisted data the live + * bridge proved against. On a pre-v3 index the `calleeIds` column is absent → + * the id query throws → ids stay empty (graceful degrade). + */ +async function sliceCalleeSetsOf(lbugPath, blockIds, exec) { + const names = new Set(); + const ids = new Set(); + if (!Array.isArray(blockIds) || blockIds.length === 0) return { names, ids }; + const nameRows = await exec( + lbugPath, + `MATCH (b:BasicBlock) WHERE b.id IN $ids RETURN b.callees AS callees`, + { ids: blockIds }, + ); + for (const r of nameRows) { + for (const n of String(r.callees ?? r[0] ?? '').split(' ')) if (n) names.add(n); + } + const idRows = await exec( + lbugPath, + `MATCH (b:BasicBlock) WHERE b.id IN $ids RETURN b.calleeIds AS calleeIds`, + { ids: blockIds }, + ).catch(() => []); + for (const r of idRows) { + for (const i of String(r.calleeIds ?? r[0] ?? '').split(' ')) if (i && i !== '*') ids.add(i); + } + return { names, ids }; +} + +/** + * U9 axis — run the resolved-id soundness gate on ONE fixture that carries an + * `idBridge` ground-truth block. Seeds `impact(mode:'pdg', line: idBridge.seedLine)`, + * extracts the id-proven statement-precise set, computes the leaf-NAME match + * counterfactual over the SAME reached depth-1 callees + the SAME exact slice + * (via the U8 `bridgeProvenSets`), and evaluates both against ground truth with + * the pure `evaluateIdBridge`. Returns the gate verdict + the realized numbers + * the report/README record (id-proven == 1 correct id; name-match == 2 = the + * over-attribution baseline). + */ +async function idBridgeAxisFor(fx, work, results, exec) { + const lbugPath = path.join(work, '.gitnexus', 'lbug'); + const pdg = results.pdg; + const idProven = idProvenIdsFromResult(pdg); + + // Name counterfactual: same depth-1 reached callees, same exact slice sets. + const reachedD1 = + pdg?.pdgInterprocedural?.byDepth?.[1] ?? pdg?.pdgInterprocedural?.byDepth?.['1'] ?? []; + const exactSlice = [ + ...(Array.isArray(pdg?.seedBlocks) ? pdg.seedBlocks : []), + ...(Array.isArray(pdg?.reachableBlocks) ? pdg.reachableBlocks : []), + ]; + const { names, ids } = await sliceCalleeSetsOf(lbugPath, exactSlice, exec); + const proven = bridgeProvenSets(reachedD1, names, ids); + const nameWouldProve = proven.nameProven.map(reachedItemKey).filter(Boolean); + const idVsName = scoreIdVsName(proven.nameProven, proven.idProven); + + const verdict = evaluateIdBridge(idProven, nameWouldProve, fx.gt.idBridge); + return { + name: fx.name, + seedLine: fx.gt.idBridge?.seedLine ?? null, + discriminating: proven.discriminating === true, + ...verdict, + idProvenCount: verdict.idProven.length, + nameProvenCount: verdict.nameProven.length, + fpEliminatedCount: idVsName.fpEliminated, + }; +} + +// ── Step 0: fixture AIS validation (gated on the live traversal; KTD9 +// circularity guard) ─────────────────────────────────────────────────────── + +/** + * Before scoring, reconcile each fixture's annotation against the LIVE analyzer: + * (a) the criterion must produce ≥1 PDG edge (no accidental no-body / cap + * truncation — a zero-edge criterion has unmeasurable ground truth); + * (b) the criterion symbol must NOT share `(filePath, startLine)` with another + * Function/Method (same-line projection ambiguity, R4) — one count query; + * (c) the annotation paths must line up with the analyzer's repo-relative + * paths (so symbol keys match across CIS/AIS). + * A fixture failing (a)/(b) is EXCLUDED from scoring and LOGGED (no silent cap). + */ +async function validateFixture(fx, work, exec) { + const lbugPath = path.join(work, '.gitnexus', 'lbug'); + // (a) criterion produces ≥1 PDG edge. Locate the criterion's blocks via the + // marker (the same technique the U6 smoke test uses) and count CDG/RD edges + // sourced inside them. + const marker = fx.gt.criterion.marker; + const blocks = await exec(lbugPath, `MATCH (b:BasicBlock) RETURN b.id AS id, b.text AS text`, {}); + const idsByAnchor = new Map(); + let anchor; + for (const b of blocks) { + const id = String(b.id ?? b[0] ?? ''); + const anc = id.slice(0, id.lastIndexOf(':')); + (idsByAnchor.get(anc) ?? idsByAnchor.set(anc, new Set()).get(anc)).add(id); + const text = String(b.text ?? b[1] ?? ''); + if (marker && text.includes(marker)) anchor = anc; + } + let critEdges = 0; + if (anchor) { + const blockIds = [...(idsByAnchor.get(anchor) ?? [])]; + if (blockIds.length > 0) { + const rows = await exec( + lbugPath, + `MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock) + WHERE r.type IN ['CDG','REACHING_DEF'] AND a.id IN $ids + RETURN count(r) AS n`, + { ids: blockIds }, + ); + critEdges = Number(rows?.[0]?.n ?? rows?.[0]?.[0] ?? 0); + } + } + + // (b) same-(filePath,startLine) collision for the criterion symbol (R4). + const collisionRows = await exec( + lbugPath, + `MATCH (s:\`Function\`) + WHERE s.name = $name AND s.filePath = $fp + RETURN s.startLine AS sl + UNION ALL + MATCH (s:\`Method\`) + WHERE s.name = $name AND s.filePath = $fp + RETURN s.startLine AS sl`, + { name: fx.gt.criterion.name, fp: fx.gt.criterion.filePath }, + ); + let sameLineCollision = false; + const startLine = collisionRows?.[0]?.sl ?? collisionRows?.[0]?.[0]; + if (startLine !== undefined && startLine !== null) { + const peers = await exec( + lbugPath, + `MATCH (s:\`Function\`) + WHERE s.filePath = $fp AND s.startLine = $sl + RETURN s.name AS name + UNION ALL + MATCH (s:\`Method\`) + WHERE s.filePath = $fp AND s.startLine = $sl + RETURN s.name AS name`, + { fp: fx.gt.criterion.filePath, sl: startLine }, + ); + sameLineCollision = (peers?.length ?? 0) > 1; + } + + const problems = []; + if (!anchor) problems.push(`criterion blocks not locatable via marker ${JSON.stringify(marker)}`); + if (critEdges === 0) + problems.push('criterion produces ZERO PDG edges (unmeasurable ground truth)'); + if (sameLineCollision) + problems.push( + 'criterion shares (filePath,startLine) with another Function/Method (R4 ambiguity)', + ); + return { critEdges, sameLineCollision, problems, measurable: problems.length === 0 }; +} + +// ── per-fixture scoring (each mode vs its NATIVE ground truth — U7 rework) ──── + +/** + * Score the CALLGRAPH mode for one fixture: its reported SYMBOL CIS against the + * fixture's `inter_AIS` (the cross-function symbols truly affected). The + * criterion symbol itself is dropped from the CIS first — callgraph never names + * the criterion as its own dependent, and `inter_AIS` is cross-function by + * construction, so a stray self-reference would be spurious noise. (In practice + * the callgraph CIS already excludes the seed; this is belt-and-suspenders.) + */ +function scoreCallgraph(gt, symbolCisKeys) { + const ais = aisByScope(gt); + const cis = new Set([...symbolCisKeys].filter((k) => k !== ais.criterionKey)); + return score(cis, ais.inter); +} + +/** + * Score the PDG mode for one fixture: its statement-LINE CIS (the + * `affectedStatements` from the line-seeded slice) against the fixture's + * `intra_AIS` LINE set. This is the intra-procedural statement-granularity + * measurement the U7 rework introduces. For an inter fixture (`intra_AIS` empty + * by design) the slice may return the router's own control-dependent statements + * — those are FPIS against the empty intra ground truth and recall is n/a, which + * is the honest "PDG is intra-procedural; on a pure-inter fixture it has no + * meaningful intra ground truth" result (symmetric to callgraph's empty intra). + */ +function scorePdg(gt, lineCisKeys) { + return score(lineCisKeys, intraLineAis(gt)); +} + +// ── U9: resolved-symbol-id soundness axis (plan 2026-06-18-001 U9; R1, R5) ──── + +/** + * Flatten the proven (statement-precise) reach of a `pdg` impact result into a + * sorted, de-duplicated id list. `pdgInterprocedural.statementPreciseByDepth` + * already holds ONLY the `callgraph-bridge` (id-proven) items across depths — + * the `unproven-bridge` callees (reached but NOT on the dependence slice) are + * dropped by `projectStatementPreciseByDepth`. So this is exactly the set the + * RESOLVED-ID bridge proves. Items without an `id` fall back to `reachedItemKey` + * (mirroring the U8 harness) so a dynamic/unresolved callee is never silently + * dropped from the gate. + */ +export function idProvenIdsFromResult(res) { + const byDepth = res?.pdgInterprocedural?.statementPreciseByDepth ?? {}; + const out = new Set(); + for (const items of Object.values(byDepth)) { + for (const it of Array.isArray(items) ? items : []) out.add(reachedItemKey(it)); + } + return [...out].filter(Boolean).sort(); +} + +/** + * PURE gate scorer for the resolved-id soundness fixture (no DB / analyze / Date / + * random — the deterministic unit test asserts this arithmetic directly). + * + * Inputs are three plain id lists derived from ONE `pdg` impact result on the + * fixture's seed line: + * - `idProven` — what the resolved-id bridge proved (statement-precise set); + * - `nameWouldProve` — what the leaf-NAME bridge would prove over the SAME + * reached items + the SAME slice (the over-attribution counterfactual); + * - `expected` — the ground-truth `idBridge` block. + * + * The gate PASSES iff the id-proven set equals `expected.idProven` EXACTLY and + * the name counterfactual strictly over-attributes (proves a superset that + * includes every `expected.fpEliminated` id). `over` is the realized + * over-attribution count the README records (= |name ∖ id|). Returns + * `{ ok, problems, idProven, nameProven, fpEliminated, over }`. + */ +export function evaluateIdBridge(idProven, nameWouldProve, expected) { + const idSet = [...new Set(idProven)].sort(); + const nameSet = [...new Set(nameWouldProve)].sort(); + const expId = [...new Set(expected?.idProven ?? [])].sort(); + const expFp = [...new Set(expected?.fpEliminated ?? [])].sort(); + const idKeys = new Set(idSet); + const over = nameSet.filter((k) => !idKeys.has(k)).sort(); + + const problems = []; + const sameSet = (a, b) => a.length === b.length && a.every((v, i) => v === b[i]); + problems.push( + ...(sameSet(idSet, expId) + ? [] + : [`id-proven set ${JSON.stringify(idSet)} != ground-truth ${JSON.stringify(expId)}`]), + ); + // The whole point: the NAME match must over-attribute (prove the collision + // callee the id match correctly drops). A name set that did NOT over-attribute + // would mean the fixture lost its discriminating power — fail loudly. + problems.push( + ...(over.length > 0 + ? [] + : ['name-match did NOT over-attribute — the fixture lost its discriminating power']), + ); + const missingFp = expFp.filter((k) => !over.includes(k)); + problems.push( + ...(missingFp.length === 0 + ? [] + : [ + `name-match did not over-attribute the expected collision id(s) ${JSON.stringify(missingFp)}`, + ]), + ); + + return { + ok: problems.length === 0, + problems, + idProven: idSet, + nameProven: nameSet, + fpEliminated: over, + over: over.length, + }; +} + +// ── U2 mutation-oracle reporting + recall-hole classification ──────────────── + +/** + * Classify a single B∖slice recall-hole line into one honest bucket so a reader is + * not misled (R2). These are NOT all "noise" — only the last is a genuine slicer + * bug, but two of the others are REAL, DOCUMENTED limitations the oracle exists to + * surface (not dismiss): + * (a) known-U1-no-ascent-gap — the missed line is the criterion function's + * continuation AFTER a call: it depends on a callee RETURN/out-param/throw + * effect the intra slice cannot ascend into without CALL_SUMMARY. A REAL, + * DOCUMENTED U1 limitation (genuinely affected, genuinely missed). + * (b) block-granularity-limit — the missed line genuinely depends on the + * criterion but is an INTERIOR statement of a coalesced straight-line CFG + * BasicBlock whose representative line already appears in the slice. The + * slice is sound at BLOCK granularity but cannot pinpoint interior statements + * at STATEMENT granularity. This is the PR's declared #1 validity threat + * (annotation-circularity / block reconciliation), here INDEPENDENTLY + * QUANTIFIED — a real limitation, not measurement noise. + * (c) oracle-direction-scope — an UPSTREAM-annotated fixture: the forward value- + * diff oracle runs in its native DOWNSTREAM sense (R1), so it cannot validate + * the reverse slice. A scope limit of the ORACLE, not a slicer property. + * (d) novel-recall-hole — anything else: a dependence the static slice missed + * that is NOT explained by (a)/(b)/(c). The strongest "real bug" signal; + * when unsure we fall back HERE (flag, never silently dismiss). + * Heuristic, line-based, conservative. + */ +function classifyRecallHole(missLine, check) { + const manual = new Set(check.manualAis ?? []); + const file = missLine.slice(0, missLine.lastIndexOf(':')); + const missNum = Number(missLine.slice(missLine.lastIndexOf(':') + 1)); + const critNum = check.criterionLine ?? null; + const sliceNums = (check.sliceLines ?? []) + .filter((k) => k.slice(0, k.lastIndexOf(':')) === file) + .map((k) => Number(k.slice(k.lastIndexOf(':') + 1))) + .filter((n) => Number.isFinite(n)); + + // (c — oracle scope limit, upstream-direction): on an UPSTREAM fixture the oracle + // runs in its native DOWNSTREAM sense (R1), so any B∖slice line is a forward-oracle + // vs reverse-slice direction mismatch — the ORACLE cannot validate this fixture, + // it is NOT a slicer recall bug. + if (check.direction === 'upstream') { + return { + bucket: 'oracle-direction-scope', + line: missLine, + reason: 'upstream fixture: forward oracle cannot validate the reverse slice', + }; + } + + // (b — real block-granularity limitation): the missed line sits BETWEEN a + // block-start that IS in the slice (or the criterion line) and the NEXT slice + // line — i.e. inside a coalesced CFG BasicBlock whose start line already appears + // in the slice. The CFG merges consecutive straight-line statements into one + // block, so an interior line can never surface as a DISTINCT slice statement even + // though it genuinely depends on the criterion. The dynamic oracle observes it at + // statement granularity, so this is the documented block-granularity limitation + // (the #1 validity threat), independently quantified — NOT measurement noise. + // Coalescing is a phenomenon of consecutive STRAIGHT-LINE statements inside one + // function (the intra stratum); on inter/mixed fixtures a "between two slice + // lines" miss is more likely a cross-function caller-continuation gap (U1), so + // restrict this classification to the intra stratum. + const blockStarts = [...sliceNums, ...(critNum !== null ? [critNum] : [])].sort((a, b) => a - b); + const startBelow = blockStarts.filter((n) => n < missNum).sort((a, b) => b - a)[0]; + const startOrSliceAtOrAbove = blockStarts.filter((n) => n >= missNum).sort((a, b) => a - b)[0]; + const shadowedByCoalescedBlock = + check.locus === 'intra' && + startBelow !== undefined && + // missLine is strictly inside [startBelow+1, nextStart-1] and the slice/criterion + // already represents that block at startBelow. + (startOrSliceAtOrAbove === undefined || startOrSliceAtOrAbove > missNum); + if (shadowedByCoalescedBlock) { + return { + bucket: 'block-granularity-limit', + line: missLine, + reason: + 'interior of a coalesced straight-line block (sound at block granularity, under-reports at statement granularity)', + }; + } + + // (a — known U1 no-ascent gap): on an inter/mixed fixture, a missed dependent line + // the slice did not reach is the classic caller-continuation-after-call gap (a + // return-value / out-param / throw effect not ascended without CALL_SUMMARY). + if (check.locus === 'inter' || check.locus === 'mixed') { + return { + bucket: 'known-U1-no-ascent-gap', + line: missLine, + reason: 'caller continuation depends on callee effect', + }; + } + + // (d — novel hole): a genuine intra dependence the static slice missed that none + // of the documented buckets explains. A miss that is ALSO a manual intra_AIS line + // is the strongest signal, but ANY unexplained intra-downstream miss is flagged + // here rather than dismissed — honesty over a clean gate. + const reason = manual.has(missLine) + ? 'manual intra_AIS line not in slice' + : 'intra dependence missed, unexplained by U1/block/direction'; + return { bucket: 'novel-recall-hole', line: missLine, reason }; +} + +// ── reporting helpers ──────────────────────────────────────────────────────── + +const fmt = (v) => (v === null || v === undefined ? 'n/a' : Number(v).toFixed(3)); +const pad = (s, n) => String(s).padEnd(n); +const lpad = (s, n) => String(s).padStart(n); + +function renderTable(strata) { + const head = + `${pad('Scope', 7)} ${pad('Mode', 10)} ${pad('Granularity', 11)} ${lpad('P', 7)} ${lpad('R', 7)} ${lpad('F1', 7)} ` + + `${lpad('|CIS|/|AIS|', 11)} ${lpad('FPIS', 6)} ${lpad('FNIS', 6)} ${lpad('n', 4)}`; + const lines = [head, '-'.repeat(head.length)]; + // PDG is scored at LINE granularity vs intra_AIS; callgraph at SYMBOL + // granularity vs inter_AIS — the column makes the "different scopes" explicit. + const gran = (mode) => (mode === 'pdg' ? 'line/intra' : 'symbol/inter'); + for (const scope of SCOPES) { + for (const mode of MODES) { + const a = strata[scope][mode]; + lines.push( + `${pad(scope, 7)} ${pad(mode, 10)} ${pad(gran(mode), 11)} ${lpad(fmt(a.precision), 7)} ${lpad(fmt(a.recall), 7)} ` + + `${lpad(fmt(a.f1), 7)} ${lpad(fmt(a.cisAisRatio), 11)} ${lpad(a.fpis, 6)} ${lpad(a.fnis, 6)} ` + + `${lpad(a.nCases, 4)}`, + ); + } + } + return lines.join('\n'); +} + +function renderUnifiedTable(unified) { + const head = + `${pad('Mode', 17)} ${pad('Axis', 13)} ${lpad('P', 7)} ${lpad('R', 7)} ${lpad('F1', 7)} ` + + `${lpad('|CIS|/|AIS|', 11)} ${lpad('FPIS', 6)} ${lpad('FNIS', 6)} ${lpad('n', 4)}`; + const lines = [head, '-'.repeat(head.length)]; + const axes = [ + ['intraLine', 'line/intra'], + ['interSymbol', 'symbol/inter'], + ]; + for (const mode of UNIFIED_MODES) { + for (const [axis, label] of axes) { + const a = unified[mode][axis]; + lines.push( + `${pad(mode, 17)} ${pad(label, 13)} ${lpad(fmt(a.precision), 7)} ${lpad(fmt(a.recall), 7)} ` + + `${lpad(fmt(a.f1), 7)} ${lpad(fmt(a.cisAisRatio), 11)} ${lpad(a.fpis, 6)} ${lpad(a.fnis, 6)} ` + + `${lpad(a.nCases, 4)}`, + ); + } + } + lines.push(''); + lines.push('Unified verdict guard: compare axes separately; do not blend line and symbol F1.'); + for (const mode of UNIFIED_MODES) { + lines.push( + ` ${mode}: min defined recall=${fmt(unified[mode].minRecall)} FPIS=${unified[mode].fpis} FNIS=${unified[mode].fnis}`, + ); + } + return lines.join('\n'); +} + +/** + * Plain-language DECISION RECOMMENDATION (F2 — the deliverable that answers + * "which is more accurate" as a verdict, not just a table). Derived from the + * measured numbers: PDG's intra-procedural STATEMENT-granularity F1 (the slice it + * is built to compute) and call-graph's inter-procedural SYMBOL-granularity F1 + * (the cross-function reach it is built to compute). + */ +function decisionRecommendation(strata, unified, underpowered, exclusions) { + // PDG is precise at intra LINE granularity; callgraph covers inter SYMBOL reach. + const pdgIntraF1 = strata.intra.pdg.f1; + const pdgIntraP = strata.intra.pdg.precision; + const pdgIntraR = strata.intra.pdg.recall; + const cgInterF1 = strata.inter.callgraph.f1; + const cgInterR = strata.inter.callgraph.recall; + const pdgMixedF1 = strata.mixed.pdg.f1; + const cgMixedF1 = strata.mixed.callgraph.f1; + + const lines = []; + lines.push('DECISION RECOMMENDATION'); + if (underpowered) { + lines.push( + `Corpus is UNDERPOWERED (below the ${FLOOR_PER_STRATUM}/stratum, ${FLOOR_TOTAL}-total floor` + + ` after exclusions) — reporting DIRECTION, not headline decimals.`, + ); + } + + // Intra-scope: the statement-anchored PDG slice — the question PDG answers. + lines.push( + `On INTRA-scope (statement granularity), the line-seeded PDG slice scores P=${fmt(pdgIntraP)} ` + + `R=${fmt(pdgIntraR)} F1=${fmt(pdgIntraF1)} against intra_AIS: it identifies the dependent ` + + `STATEMENTS of the changed line precisely. Call-graph mode cannot resolve below function ` + + `granularity, so on a self-contained function it names no other symbol (intra recall n/a — ` + + `no cross-function truth to find). PDG is the engine for "which statements does this line affect?".`, + ); + + // Inter-scope: the cross-function blast radius — the question call-graph answers. + lines.push( + `On INTER-scope (symbol granularity), call-graph scores R=${fmt(cgInterR)} F1=${fmt(cgInterF1)} ` + + `against inter_AIS: it recovers the cross-function callees exactly. Unified PDG mode now ` + + `attaches the same inter-symbol reach in interproceduralByDepth/byDepth while keeping statement reach in ` + + `affectedStatements, so the symbol axis can be compared directly against callgraph.`, + ); + + // Mixed-scope: both engines contribute, each in its own scope. + if (pdgMixedF1 !== null || cgMixedF1 !== null) { + lines.push( + `On MIXED-scope, the two are COMPLEMENTARY: PDG resolves the intra statement set ` + + `(F1=${fmt(pdgMixedF1)} vs intra_AIS) while call-graph reaches the callee(s) ` + + `(F1=${fmt(cgMixedF1)} vs inter_AIS). Neither alone covers the full mixed blast radius.`, + ); + } + + if (unified) { + lines.push( + `Unified-axis check: current callgraph leaves the intra-line axis empty. Unified PDG now ` + + `covers both axes, while composed-current remains the control baseline that combines the ` + + `standalone callgraph symbol reach with PDG statement reach. composed-current reaches min ` + + `defined recall=${fmt(unified['composed-current'].minRecall)} with ` + + `FPIS=${unified['composed-current'].fpis} and FNIS=${unified['composed-current'].fnis}; ` + + `pdg should match that recall before any default-switch discussion.`, + ); + } + + lines.push( + `VERDICT: keep option-driven comparison, but mode:'pdg' is now the unified PDG-facing ` + + `answer: statement-level affectedStatements come from the persisted CDG/REACHING_DEF slice, ` + + `and inter-procedural symbol reach is carried in interproceduralByDepth/byDepth. mode:'callgraph' remains the ` + + `default/comparator for the established symbol-only traversal. The accuracy decision should be ` + + `made from the unified axes: pdg must preserve statement recall while matching the composed ` + + `inter-symbol baseline and bounding FPIS.`, + ); + if (exclusions.length > 0) { + lines.push( + `Excluded from scoring: ${exclusions.map((e) => `${e.name} (${e.reason})`).join('; ')}.`, + ); + } + return lines.join('\n'); +} + +// ── main run ───────────────────────────────────────────────────────────────── + +async function run() { + const CHECK = process.argv.includes('--check'); + const JSON_OUT = process.argv.includes('--json'); + // U2 dynamic-oracle (opt-in, BENCH-ADDITIVE). --mutation runs the value-diff + // forward-slice oracle and prints per-fixture recall + circularity rows; + // --mutation-strict would later flip Gate 4 to a hard exit (report-only now). + const MUTATION = process.argv.includes('--mutation'); + const MUTATION_STRICT = process.argv.includes('--mutation-strict'); + // Optional subset for a fast substrate proof: --only=a,b,c or GN_IMPACT_PDG_ONLY=a,b + const onlyArg = process.argv.find((a) => a.startsWith('--only=')); + const onlyEnv = process.env.GN_IMPACT_PDG_ONLY; + const filter = (onlyArg ? onlyArg.slice('--only='.length) : onlyEnv || '') + .split(',') + .map((s) => s.trim()) + .filter(Boolean); + + const fixtures = loadFixtures(filter.length ? filter : null); + if (fixtures.length === 0) throw new Error('no fixtures found'); + + // K repeats for substrate-stability (F5). --check runs K times and gates on + // the per-(mode,scope) MEDIAN F1, so a flaky analyze edge cannot trip the band. + const K = CHECK ? Number(process.env.GN_IMPACT_PDG_K || 1) : 1; + + const home = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-impact-pdg-home-')); + const { initLbug, executeParameterized, closeLbug } = await import( + path.join(REPO_ROOT, 'src', 'core', 'lbug', 'pool-adapter.ts') + ); + + // exec wrapper that ensures the pool is initialised for Step 0's raw queries. + const initialised = new Set(); + const exec = async (lbugPath, q, p) => { + if (!initialised.has(lbugPath)) { + await initLbug(lbugPath, lbugPath).catch(() => {}); + initialised.add(lbugPath); + } + return executeParameterized(lbugPath, q, p); + }; + + const exclusions = []; + const perRunStrata = []; // K runs × { scope: { mode: aggregate } } + const perRunUnified = []; // K runs × { mode: { intraLine, interSymbol } } + let perCaseDetail = null; // last run's per-case detail for the report + let degradedCheck = null; + const idBridgeChecks = []; // U9 resolved-id soundness axis (one per idBridge fixture) + const mutationChecks = []; // U2 dynamic-oracle rows (one per measurable fixture, --mutation only) + + try { + for (let runIdx = 0; runIdx < K; runIdx++) { + // perCaseScores[scope][mode] = array of per-fixture score objects + const perScopeMode = {}; + for (const s of SCOPES) { + perScopeMode[s] = {}; + for (const m of MODES) perScopeMode[s][m] = []; + } + const detail = []; + const perUnifiedMode = {}; + for (const m of UNIFIED_MODES) perUnifiedMode[m] = []; + + for (const fx of fixtures) { + if (fx.excluded) { + if (runIdx === 0) + exclusions.push({ name: fx.name, reason: 'no-body (pdgScoring:exclude / KTD6)' }); + continue; + } + const { work, results } = await analyzeAndImpact(fx, home, { pdgOn: true }); + try { + // Step 0 — reconcile annotation against the live traversal. + const v = await validateFixture(fx, work, exec); + if (!v.measurable) { + if (runIdx === 0) exclusions.push({ name: fx.name, reason: v.problems.join(' + ') }); + continue; + } + + // CALLGRAPH: symbol CIS vs inter_AIS. PDG: line CIS vs intra_AIS. + const cg = callgraphCisFromResult(results.callgraph); + const pdg = pdgCisFromResult(results.pdg); + const cgScore = scoreCallgraph(fx.gt, cg.keys); // symbol/inter + const pdgScore = scorePdg(fx.gt, pdg.intraLineKeys); // line/intra (FU-A: intra-tagged only) + + const unifiedTruth = unifiedAis(fx.gt); + const cgUnified = callgraphUnifiedCis(fx.gt, cg.keys); + const pdgUnified = pdgUnifiedCis(pdg.intraLineKeys, pdg.symbolKeys, fx.gt); + const composedUnified = composeUnifiedCis(cgUnified, pdgUnified); + const unifiedScores = { + callgraph: scoreUnifiedAxes(cgUnified, unifiedTruth), + pdg: scoreUnifiedAxes(pdgUnified, unifiedTruth), + 'composed-current': scoreUnifiedAxes(composedUnified, unifiedTruth), + }; + for (const m of UNIFIED_MODES) perUnifiedMode[m].push(unifiedScores[m]); + + const locusScope = fx.gt.locus; // the stratum this fixture belongs to + // A fixture is scored in its OWN locus stratum (intra/inter/mixed), + // each mode against its native ground truth (symbol vs line). + if (SCOPES.includes(locusScope)) { + perScopeMode[locusScope].callgraph.push(cgScore); + perScopeMode[locusScope].pdg.push(pdgScore); + } + + if (runIdx === 0) { + detail.push({ + name: fx.name, + locus: fx.gt.locus, + criterion: fx.gt.criterion.name, + direction: fx.gt.criterion.direction, + criterionLine: fx.gt.criterion.line ?? null, + critEdges: v.critEdges, + cg: { + count: results.callgraph.impactedCount, + symbols: [...cg.keys].sort(), + score: cgScore, // vs inter_AIS (symbol) + }, + pdg: { + affectedStatementCount: pdg.meta.affectedStatementCount, + blockCount: pdg.meta.blockCount, + criterionLine: pdg.meta.criterionLine, + lines: [...pdg.lineKeys].sort(), + symbols: [...pdg.symbolKeys].sort(), + score: pdgScore, // vs intra_AIS (line) + }, + unified: unifiedScores, + }); + } + } finally { + await closeLbug(path.join(work, '.gitnexus', 'lbug')).catch(() => {}); + initialised.delete(path.join(work, '.gitnexus', 'lbug')); + fs.rmSync(work, { recursive: true, force: true }); + } + } + + // Aggregate this run's native strata and unified two-axis comparison. + const strata = {}; + for (const s of SCOPES) { + strata[s] = {}; + for (const m of MODES) strata[s][m] = aggregate(perScopeMode[s][m]); + } + const unified = {}; + for (const m of UNIFIED_MODES) unified[m] = aggregateUnifiedScores(perUnifiedMode[m]); + perRunStrata.push(strata); + perRunUnified.push(unified); + if (runIdx === 0) perCaseDetail = detail; + } + + // ── Degraded-index check (KTD7): on ONE intra fixture, analyze WITHOUT + // --pdg and assert PDG mode reports a degradation note (skipped, not 0/0). + const degTarget = fixtures.find((f) => !f.excluded && f.gt.locus === 'intra'); + if (degTarget) { + const { work, results } = await analyzeAndImpact(degTarget, home, { pdgOn: false }); + try { + const pdgRes = results.pdg; + degradedCheck = { + name: degTarget.name, + pdgLayer: pdgRes.pdgLayer ?? null, + note: (pdgRes.note ?? pdgRes.error ?? '').slice(0, 140), + skipped: pdgRes.pdgLayer !== undefined && pdgRes.pdgLayer !== 'ready', + }; + } finally { + const lbugPath = path.join(work, '.gitnexus', 'lbug'); + await closeLbug(lbugPath).catch(() => {}); + initialised.delete(lbugPath); + fs.rmSync(work, { recursive: true, force: true }); + } + } + + // ── U9 resolved-id soundness axis (R1, R5): for each fixture carrying an + // `idBridge` ground-truth block, prove that the resolved-ID bridge labels + // EXACTLY the right callee statement-precise while the leaf-NAME bridge would + // over-attribute the same-named collision callee. analyzeAndImpact already + // seeds `results.pdg` on `criterion.line` (== idBridge.seedLine), so the + // statement-precise set is the id-proven set for the seed line directly. + for (const fx of fixtures.filter((f) => f.gt.idBridge)) { + const { work, results } = await analyzeAndImpact(fx, home, { pdgOn: true }); + try { + idBridgeChecks.push(await idBridgeAxisFor(fx, work, results, exec)); + } finally { + const lbugPath = path.join(work, '.gitnexus', 'lbug'); + await closeLbug(lbugPath).catch(() => {}); + initialised.delete(lbugPath); + fs.rmSync(work, { recursive: true, force: true }); + } + } + + // ── U2 dynamic-oracle mutation pass (opt-in --mutation; BENCH-ADDITIVE) ── + // For each fixture: re-analyze into the SAME temp working copy, take the SAME + // live static PDG slice the F1 metric scores, derive the behavioral (dynamic + // forward) AIS by value-diff over line-scoped mutants on the criterion line, + // then score mutation_recall vs the slice and the circularity diff vs the + // manual intra_AIS. Runs ONCE (not K times) — the oracle is deterministic and + // the value-diff is the load-bearing signal, not substrate-noise-prone like F1. + if (MUTATION) { + // LAZY-load the oracle (+ its heavy @babel/* deps) only when --mutation is + // actually requested — see the import-block note above for why this must NOT + // be a static top-level import. + const { deriveBehavioralAis, writeMutationSidecar } = await import('./mutation-oracle.mjs'); + for (const fx of fixtures) { + // nobody-interface-excluded has NO body → no statement to seed/mutate → + // oracle-excluded (not a silent skip; printed in the row). + const noBody = fx.gt.locus === 'n/a' || !fx.gt.criterion.line; + // R1: the two direction:'upstream' fixtures are a forward-oracle mismatch. + // Run the oracle in its native downstream sense + the circularity cross- + // check, but DO NOT apply the recall gate to them (oracle-direction-excluded). + const upstream = fx.gt.criterion.direction === 'upstream'; + // intra-overloaded-callee (pdgScoring:exclude, has a body) is corroboration + // for the id-discrimination axis, NOT an AIS recall case. + const idCorroboration = fx.gt.pdgScoring === 'exclude' && !noBody; + + if (noBody) { + mutationChecks.push({ + name: fx.name, + locus: fx.gt.locus, + direction: fx.gt.criterion.direction, + criterionKey: null, + behavioralAis: [], + sliceLines: [], + manualAis: [], + recall: null, + recallGated: false, + scopeNote: 'oracle-excluded: no body (no statement to mutate)', + mutants: [], + circularity: { beyondManual: [], confirmed: [], manualOnly: [] }, + skipped: 'no-body', + }); + continue; + } + + const { work, results } = await analyzeAndImpact(fx, home, { pdgOn: true }); + try { + // The SAME live static slice the F1 metric scores. + const slice = pdgLineCis(results.pdg?.affectedStatements); + const manualAis = intraLineAis(fx.gt); + const derived = await deriveBehavioralAis(fx, work); + writeMutationSidecar(fx, derived); + + const B = new Set(derived.behavioralAis); + const rec = mutationRecall(B, slice); + const circ = circularityDiff(B, manualAis); + // The recall gate applies to DOWNSTREAM, non-corroboration fixtures only. + const recallGated = !upstream && !idCorroboration; + const scopeNote = idCorroboration + ? 'id-discrimination corroboration (excluded from recall gate)' + : upstream + ? 'oracle-direction-excluded: upstream fixture (forward-oracle native downstream)' + : null; + + mutationChecks.push({ + name: fx.name, + locus: fx.gt.locus, + direction: fx.gt.criterion.direction, + criterionKey: derived.criterionKey, + criterionLine: derived.criterionLine, + behavioralAis: derived.behavioralAis, + sliceLines: [...slice].sort(), + manualAis: [...manualAis].sort(), + recall: rec.recall, + recallGated, + intersection: rec.intersection, + bSize: rec.bSize, + sliceSize: rec.sliceSize, + missing: rec.missing, // B ∖ slice — recall hole + extra: rec.extra, // slice ∖ B — sound over-approx (informational) + circularity: circ, + scopeNote, + paramTypes: derived.paramTypes, + inputs: derived.inputs, + mutants: derived.mutants, + skipped: derived.skipped, + }); + } finally { + const lbugPath = path.join(work, '.gitnexus', 'lbug'); + await closeLbug(lbugPath).catch(() => {}); + initialised.delete(lbugPath); + fs.rmSync(work, { recursive: true, force: true }); + } + } + } + } finally { + fs.rmSync(home, { recursive: true, force: true }); + } + + // ── Collapse K runs into the report strata: per (mode,scope) take the MEDIAN + // F1 across runs (F5 substrate stability); other fields from run 0. + const strata0 = perRunStrata[0]; + const report = {}; + for (const s of SCOPES) { + report[s] = {}; + for (const m of MODES) { + const f1s = perRunStrata.map((r) => r[s][m].f1).filter((v) => v !== null && v !== undefined); + const pmeds = perRunStrata + .map((r) => r[s][m].precision) + .filter((v) => v !== null && v !== undefined); + const rmeds = perRunStrata + .map((r) => r[s][m].recall) + .filter((v) => v !== null && v !== undefined); + report[s][m] = { + ...strata0[s][m], + f1: f1s.length ? median(f1s) : null, + precision: pmeds.length ? median(pmeds) : null, + recall: rmeds.length ? median(rmeds) : null, + }; + } + } + + const unified0 = perRunUnified[0]; + const unifiedReport = {}; + for (const mode of UNIFIED_MODES) { + unifiedReport[mode] = { ...unified0[mode] }; + for (const axis of ['intraLine', 'interSymbol']) { + const f1s = perRunUnified + .map((r) => r[mode][axis].f1) + .filter((v) => v !== null && v !== undefined); + const pmeds = perRunUnified + .map((r) => r[mode][axis].precision) + .filter((v) => v !== null && v !== undefined); + const rmeds = perRunUnified + .map((r) => r[mode][axis].recall) + .filter((v) => v !== null && v !== undefined); + unifiedReport[mode][axis] = { + ...unified0[mode][axis], + f1: f1s.length ? median(f1s) : null, + precision: pmeds.length ? median(pmeds) : null, + recall: rmeds.length ? median(rmeds) : null, + }; + } + const minRecalls = perRunUnified + .map((r) => r[mode].minRecall) + .filter((v) => v !== null && v !== undefined); + unifiedReport[mode].minRecall = minRecalls.length ? median(minRecalls) : null; + } + + // Underpowered floor (F3): measured cases per stratum after exclusions. + const measurableTotal = SCOPES.reduce( + (a, s) => a + Math.max(report[s].callgraph.nCases, report[s].pdg.nCases), + 0, + ); + const underpowered = + measurableTotal < FLOOR_TOTAL || + SCOPES.some( + (s) => Math.max(report[s].callgraph.nCases, report[s].pdg.nCases) < FLOOR_PER_STRATUM, + ); + + const annotationFingerprint = fingerprintAnnotationSet(fixtures, sha256); + + const machineReport = { + analyzerVersion: JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf8')) + .version, + corpus: { total: fixtures.length, measurable: measurableTotal, excluded: exclusions }, + underpowered, + floor: { perStratum: FLOOR_PER_STRATUM, total: FLOOR_TOTAL }, + strata: report, + unified: unifiedReport, + perCase: perCaseDetail, + degradedCheck, + idBridgeChecks, + mutation: MUTATION + ? { + fingerprint: fingerprintMutationSet(mutationChecks, sha256), + checks: mutationChecks, + } + : null, + annotationFingerprint, + runsK: K, + }; + + // ── output ────────────────────────────────────────────────────────────── + if (JSON_OUT) { + process.stdout.write(JSON.stringify(machineReport, null, 2) + '\n'); + } else { + const out = []; + out.push('=== impact-PDG accuracy report ==='); + out.push( + `analyzer ${machineReport.analyzerVersion} | corpus ${fixtures.length} ` + + `(${measurableTotal} measurable, ${exclusions.length} excluded) | runs K=${K}`, + ); + out.push(''); + out.push( + 'Stratified P/R/F1 (PDG: line granularity vs intra_AIS; callgraph: symbol vs inter_AIS):', + ); + out.push(renderTable(report)); + out.push(''); + out.push('Unified impact axes (additive; native table above is unchanged):'); + out.push(renderUnifiedTable(unifiedReport)); + out.push(''); + out.push( + 'Per-case: PDG slice (line/intra) and callgraph reach (symbol/inter), with FPIS/FNIS:', + ); + for (const d of perCaseDetail) { + out.push( + ` ${pad(d.name, 28)} locus=${pad(d.locus, 6)} line=${lpad(d.criterionLine ?? '-', 3)}`, + ); + // PDG line slice: F1 vs intra_AIS, with the false-positive / false-negative lines. + const ps = d.pdg.score; + out.push( + ` pdg line/intra : P=${fmt(ps.precision)} R=${fmt(ps.recall)} F1=${fmt(ps.f1)} ` + + `|CIS|=${d.pdg.affectedStatementCount} blocks=${d.pdg.blockCount} ` + + `FPIS=${ps.fpisCount} FNIS=${ps.fnisCount}`, + ); + if (ps.fpisCount > 0) out.push(` FPIS(noise): ${ps.fpis.join(', ')}`); + if (ps.fnisCount > 0) out.push(` FNIS(missed): ${ps.fnis.join(', ')}`); + // Callgraph symbol reach: F1 vs inter_AIS. + const cs = d.cg.score; + out.push( + ` cg symbol/inter: P=${fmt(cs.precision)} R=${fmt(cs.recall)} F1=${fmt(cs.f1)} ` + + `|CIS|=${d.cg.count} FPIS=${cs.fpisCount} FNIS=${cs.fnisCount}`, + ); + if (cs.fnisCount > 0) out.push(` FNIS(missed): ${cs.fnis.join(', ')}`); + } + out.push(''); + if (degradedCheck) { + out.push( + `Degraded-index probe (KTD7): ${degradedCheck.name} analyzed WITHOUT --pdg → ` + + `pdgLayer=${degradedCheck.pdgLayer} skipped=${degradedCheck.skipped}`, + ); + out.push(` note: ${degradedCheck.note}`); + out.push(''); + } + if (idBridgeChecks.length > 0) { + out.push('Resolved-id soundness axis (U9 — R1/R5): id-match proves the right callee,'); + out.push('name-match over-attributes the same-leaf-name collision callee:'); + for (const c of idBridgeChecks) { + out.push( + ` ${pad(c.name, 28)} seed=${lpad(c.seedLine ?? '-', 3)} ` + + `id-proven=${c.idProvenCount} name-would-prove=${c.nameProvenCount} ` + + `(over-attribution=${c.over}) ${c.ok ? 'PASS' : 'FAIL'}`, + ); + out.push(` id-proven : ${c.idProven.join(', ') || '(none)'}`); + out.push(` eliminated: ${c.fpEliminated.join(', ') || '(none)'}`); + out.push(...(c.problems.length > 0 ? [` problems : ${c.problems.join('; ')}`] : [])); + } + out.push(''); + } + if (MUTATION) { + out.push( + 'U2 dynamic-oracle (value-diff forward slice; Agrawal-Horgan/Tip/Voas): mutation_recall =', + ); + out.push( + '|B ∩ slice|/|B| (B = dynamic AIS the oracle PROVED). circularity = B ∖ manual_intra_AIS', + ); + out.push( + '(non-empty ⇒ the manual annotation MISSED a real dependence — independent evidence):', + ); + const head = + ` ${pad('Case', 28)} ${pad('locus', 6)} ${pad('dir', 10)} ${lpad('|B|', 4)} ` + + `${lpad('|slice|', 8)} ${lpad('recall', 7)} ${lpad('gated', 6)} ${lpad('circ', 5)}`; + out.push(head); + out.push(' ' + '-'.repeat(head.length - 2)); + for (const c of mutationChecks) { + out.push( + ` ${pad(c.name, 28)} ${pad(c.locus, 6)} ${pad(c.direction, 10)} ${lpad(c.bSize ?? 0, 4)} ` + + `${lpad(c.sliceSize ?? 0, 8)} ${lpad(fmt(c.recall), 7)} ${lpad(c.recallGated ? 'yes' : 'no', 6)} ` + + `${lpad((c.circularity?.beyondManual ?? []).length, 5)}`, + ); + out.push(...(c.scopeNote ? [` scope: ${c.scopeNote}`] : [])); + out.push(...(c.skipped ? [` oracle-skipped: ${c.skipped}`] : [])); + out.push( + ...((c.missing ?? []).length > 0 + ? [` B∖slice (recall hole): ${c.missing.join(', ')}`] + : []), + ); + // B∖manual is only a circularity SIGNAL on intra fixtures (whose intra_AIS + // claims completeness). On inter/mixed fixtures intra_AIS deliberately omits + // cross-function lines, so B∖manual there is EXPECTED (callee-body lines), + // not an annotation miss — labelled accordingly so it is not misread. + const circMeaningful = c.locus === 'intra'; + out.push( + ...((c.circularity?.beyondManual ?? []).length > 0 + ? [ + ` B∖manual (${circMeaningful ? 'WARN — intra annotation missed' : 'expected: cross-function, intra_AIS empty-by-design'}): ${c.circularity.beyondManual.join(', ')}`, + ] + : []), + ); + } + out.push(''); + // Classify EVERY recall<1.0 (R2) so a reader is not misled. + const holes = []; + for (const c of mutationChecks) { + for (const miss of c.missing ?? []) + holes.push({ ...classifyRecallHole(miss, c), case: c.name }); + } + out.push('Recall-hole classification (R2 — every B∖slice line):'); + if (holes.length === 0) { + out.push(' (none — every gated fixture has mutation_recall == 1.0)'); + } else { + for (const h of holes) out.push(` [${h.bucket}] ${h.case}: ${h.line} — ${h.reason}`); + } + // Corpus circularity headline — meaningful ONLY on intra fixtures (whose + // intra_AIS claims completeness). A non-empty B∖manual there would be the + // headline independent evidence that the hand annotation is incomplete. + const intraCirc = mutationChecks.filter( + (c) => c.locus === 'intra' && (c.circularity?.beyondManual ?? []).length > 0, + ); + out.push( + intraCirc.length > 0 + ? `CIRCULARITY: on ${intraCirc.length} intra fixture(s) the oracle proved a dependence the ` + + `manual intra_AIS missed: ${intraCirc.map((c) => `${c.name} [${c.circularity.beyondManual.join(', ')}]`).join('; ')}. ` + + `Classify each (block-granularity reconciliation vs genuine annotation gap) before acting.` + : 'CIRCULARITY: clean on the intra stratum — the dynamic oracle proved no INTRA dependence the ' + + 'manual intra_AIS missed (inter/mixed B∖manual is expected cross-function reach, not a miss).', + ); + out.push(`Mutation fingerprint: ${fingerprintMutationSet(mutationChecks, sha256)}`); + out.push(''); + } + out.push(`Annotation fingerprint: ${annotationFingerprint}`); + out.push(''); + out.push(decisionRecommendation(report, unifiedReport, underpowered, exclusions)); + process.stdout.write(out.join('\n') + '\n'); + } + + // ── --check: two gates (KTD10) + F5 substrate stability ─────────────────── + if (CHECK) { + if (!fs.existsSync(BASELINE_PATH)) { + process.stderr.write(`[impact-pdg --check] FAIL: no baselines.json at ${BASELINE_PATH}\n`); + process.exit(1); + } + const baselines = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8')); + const failures = []; + + // Gate 1 — order-independent annotation fingerprint (unreviewed GT edits). + if (baselines.annotationFingerprint !== annotationFingerprint) { + failures.push( + `annotation fingerprint drift: ground-truth set changed without re-baseline ` + + `(got ${annotationFingerprint}, expected ${baselines.annotationFingerprint}) — ` + + `review the ground-truth.json edits, then re-baseline.`, + ); + } + + // Gate 2 — one-sided F1 regression band per mode per scope (improvements + // pass freely; only a DROP beyond ε fails). Median-of-K already applied. + const eps = baselines.epsilon ?? 0.05; + const bands = baselines.f1Bands ?? {}; + for (const s of SCOPES) { + for (const m of MODES) { + const baseF1 = bands[s]?.[m]; + const gotF1 = report[s][m].f1; + if (baseF1 === undefined || baseF1 === null) continue; // no band ⇒ nothing to regress against + if (gotF1 === null) { + // F1 became undefined where a baseline existed — a structural change + // (the scope lost all measurable cases). Flag it, don't pass silently. + failures.push( + `${s}/${m}: F1 is now n/a but baseline was ${fmt(baseF1)} (scope lost measurable cases?)`, + ); + continue; + } + if (gotF1 < baseF1 - eps) { + failures.push( + `${s}/${m}: F1 ${fmt(gotF1)} < baseline ${fmt(baseF1)} − ε(${eps}) = ${fmt(baseF1 - eps)} ` + + `(median of K=${K})`, + ); + } + } + } + + // Gate 3 — resolved-id soundness axis (U9, R1/R5): every fixture carrying an + // `idBridge` block must PASS its gate (id-proven == the single correct id AND + // the name match strictly over-attributes the eliminated collision id). The + // expected sets live in the ground-truth `idBridge` block (covered by Gate 1's + // fingerprint), so this gate has no separate baseline number to drift. + const idBridgeFixtureCount = fixtures.filter((f) => f.gt.idBridge).length; + for (const c of idBridgeChecks) { + failures.push( + ...(c.ok ? [] : [`id-bridge ${c.name} (seed ${c.seedLine}): ${c.problems.join('; ')}`]), + ); + } + // A declared idBridge fixture that produced NO check (analyze/seed dropout) + // must fail loudly — never let the soundness gate silently vanish. + failures.push( + ...(idBridgeChecks.length === idBridgeFixtureCount + ? [] + : [ + `id-bridge axis ran ${idBridgeChecks.length} of ${idBridgeFixtureCount} declared ` + + `idBridge fixtures (a fixture dropped out of the gate)`, + ]), + ); + + // Gate 4 — U2 mutation recall (REPORT-ONLY this landing; --mutation only). + // mutation_recall < 1.0 on a GATED (downstream, non-corroboration) fixture + // means the dynamic oracle proved a dependence the static slice missed. We + // PRINT the gate line + the numbers but DO NOT process.exit(1) yet — flipping + // to a hard gate later is a one-flag change (--mutation-strict already wired). + if (MUTATION) { + const gatedHoles = mutationChecks.filter( + (c) => c.recallGated && c.recall !== null && c.recall < 1, + ); + const gatedCount = mutationChecks.filter((c) => c.recallGated).length; + // A future hard gate should fire only on NOVEL holes. The documented, + // expected buckets are excluded: known-U1-no-ascent-gap (return/effect ascent + // U1 lacks), block-granularity-limit (interior of a coalesced block — the + // slice is sound at block granularity), and oracle-direction-scope (upstream + // fixture the forward oracle cannot validate). Compute the novel subset via + // the same classifier the report uses. + const novelHoles = []; + for (const c of gatedHoles) { + for (const miss of c.missing ?? []) { + const cls = classifyRecallHole(miss, c); + if (cls.bucket === 'novel-recall-hole') novelHoles.push(`${c.name}:${miss}`); + } + } + const allRecall1 = + gatedHoles.length === 0 + ? 'all gated recall==1.0' + : gatedHoles + .map((c) => `${c.name} recall=${fmt(c.recall)} (B∖slice: ${c.missing.join(', ')})`) + .join('; '); + process.stderr.write( + `[impact-pdg --check] Gate 4 (mutation recall, REPORT-ONLY): ` + + `${gatedCount} gated fixture(s), ${gatedHoles.length} below 1.0 ` + + `(${novelHoles.length} NOVEL after classification) — ${allRecall1}\n`, + ); + // --mutation-strict opt-in: flip the report-only gate to a hard failure on + // NOVEL holes only (one-flag change to make this the live gate later). + failures.push( + ...(MUTATION_STRICT && novelHoles.length > 0 + ? [`mutation recall: NOVEL hole(s) (--mutation-strict): ${novelHoles.join(', ')}`] + : []), + ); + } + + if (failures.length > 0) { + for (const f of failures) process.stderr.write(`[impact-pdg --check] FAIL: ${f}\n`); + process.exit(1); + } + process.stderr.write( + `[impact-pdg --check] PASS (${SCOPES.length} scopes × ${MODES.length} modes, ` + + `fingerprint OK, ${idBridgeChecks.length} id-bridge fixture(s) sound, K=${K})\n`, + ); + } +} + +// Only execute the harness when invoked as the CLI entrypoint — NOT when this +// module is imported (e.g. impact-pdg-id-bridge-gate.test.ts imports the exported +// pure helpers). Importing must be side-effect-free: an unguarded run() kicks off +// the full real-analyze report in the background, which under the full vitest +// suite leaks an unhandled error that Vitest attributes to the importing file. +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + run().catch((err) => { + process.stderr.write(`[impact-pdg] ERROR: ${err?.stack || err}\n`); + process.exit(1); + }); +} diff --git a/gitnexus/bench/impact-pdg/metrics.mjs b/gitnexus/bench/impact-pdg/metrics.mjs new file mode 100644 index 000000000..d61d0f017 --- /dev/null +++ b/gitnexus/bench/impact-pdg/metrics.mjs @@ -0,0 +1,498 @@ +/** + * Pure scorer + annotation canonicalizer for the impact-PDG accuracy harness + * (U7). NO substrate here — no `runPipelineFromRepo`, no `LocalBackend`, no DB, + * no child-process `analyze`. Everything in this module is a pure function over + * plain symbol-set inputs, so the metric-math unit test + * (`test/unit/impact-pdg-metric-math.test.ts`) can import and assert the + * arithmetic deterministically, staying OUT of the flaky full-pipeline lane + * (Arch-review Issue 5). `measure.mjs` imports these for the live loop. + * + * ── CIS / AIS framing (KTD9 — Arnold–Bohner) ─────────────────────────────── + * CIS = Computed Impact Set: what a mode REPORTS as impacted. + * AIS = Actual Impact Set: the curated ground-truth set truly affected. + * precision = |AIS∩CIS| / |CIS| (over-approximation cost; ∅ CIS ⇒ undefined) + * recall = |AIS∩CIS| / |AIS| (under-approximation; ∅ AIS ⇒ undefined) + * F1 = harmonic mean (undefined if either is undefined) + * FPIS = CIS − AIS (false positives — noise) + * FNIS = AIS − CIS (false negatives — the DANGEROUS miss for a safety tool) + * + * ── Granularity: the two engines measure DIFFERENT scopes (U7 rework) ─────── + * The two `impact` engines answer different questions at different granularities, + * so they are scored against different ground truths: + * + * - **PDG mode** (`impact({mode:'pdg', line:N})`) is scored at intra-procedural + * STATEMENT granularity. The statement-anchored slice returns + * `affectedStatements: {line,filePath,text}[]` — the dependent statements of + * the criterion line N. CIS_pdg is the set of those LINE keys + * (`:`, via `pdgLineCis`); AIS is the `intra_AIS` line set + * (via `intraLineAis`). This is the unit at which PDG is precise. + * - **Call-graph mode** is scored at inter-procedural SYMBOL granularity. CIS is + * the reported symbols (`@`, via `symbolKey`); AIS is the + * `inter_AIS` symbol set (via `aisByScope`). This is the unit at which the + * call-graph blast radius is meaningful. + * + * Neither native row is a strict refinement of the other: the PDG native + * metric resolves dependent statements WITHIN a function, while call-graph + * resolves cross-function symbol reach. The unified axes report checks whether + * mode:'pdg' carries both outputs without blending their granularities. + * + * `partitionCisByScope`/`aisByScope` (symbol-level) remain for the call-graph + * path; `pdgLineCis`/`intraLineAis` (line-level) drive the PDG path. + */ + +/** Order-independent symbol key. Collapses statement lines onto their symbol. */ +export function symbolKey(symbol, filePath) { + return `${symbol}@${filePath}`; +} + +/** + * Statement-LINE key for the PDG path (U7 rework). PDG mode is now scored at + * intra-procedural STATEMENT granularity: the `impact({mode:'pdg', line})` slice + * returns `affectedStatements: {line, filePath, text}[]`, and the intra ground + * truth is the per-line `intra_AIS`. A line key is `:` — + * order-independent and statement-granular (NOT collapsed onto the owning + * symbol the way `symbolKey` is). This is the unit at which PDG is precise. + */ +export function lineKey(filePath, line) { + return `${filePath}:${line}`; +} + +/** + * Unified-impact key spaces keep the two granularities explicit. A tagged key + * is never compared across axes: `statement:src/a.ts:10` and + * `symbol:handler@src/a.ts` are different measurement units by design. + */ +export function unifiedLineKey(filePath, line) { + return `statement:${lineKey(filePath, line)}`; +} + +export function unifiedSymbolKey(symbol, filePath) { + return `symbol:${symbolKey(symbol, filePath)}`; +} + +/** + * CIS_pdg = the set of affected-statement LINE keys from an impact pdg result. + * + * `scope` (FU-A) ∈ `undefined | 'intra' | 'inter'`: + * - `undefined` → the all-union over the full slice (back-compat; the + * metric-math unit test and the U2 mutation-oracle rely on this form). + * - `'intra'` / `'inter'` → keep only statements carrying that `scope` tag, so + * U1's cross-function reach stops landing on the intra-line axis as FPIS. + */ +export function pdgLineCis(affectedStatements, scope) { + const out = new Set(); + for (const s of affectedStatements ?? []) { + if (scope !== undefined && s?.scope !== scope) continue; + if (s && typeof s.line === 'number' && typeof s.filePath === 'string') { + out.add(lineKey(s.filePath, s.line)); + } + } + return out; +} + +/** AIS_intra = the set of `intra_AIS` LINE keys (statement-granular ground truth). */ +export function intraLineAis(gt) { + const out = new Set(); + for (const e of gt.intra_AIS ?? []) { + if (e && typeof e.line === 'number' && typeof e.filePath === 'string') { + out.add(lineKey(e.filePath, e.line)); + } + } + return out; +} + +/** Unified AIS = two explicit axes, never one blended line+symbol set. */ +export function unifiedAis(gt) { + const intraLine = new Set(); + for (const e of gt.intra_AIS ?? []) { + if (e && typeof e.line === 'number' && typeof e.filePath === 'string') { + intraLine.add(unifiedLineKey(e.filePath, e.line)); + } + } + + const interSymbol = new Set(); + for (const e of gt.inter_AIS ?? []) { + if (e && typeof e.symbol === 'string' && typeof e.filePath === 'string') { + interSymbol.add(unifiedSymbolKey(e.symbol, e.filePath)); + } + } + + return { intraLine, interSymbol }; +} + +/** Canonicalize an iterable of {symbol,filePath} (or pre-made keys) → a Set. */ +export function toKeySet(entries) { + const out = new Set(); + for (const e of entries) { + if (typeof e === 'string') out.add(e); + else out.add(symbolKey(e.symbol, e.filePath)); + } + return out; +} + +function intersectionSize(a, b) { + let n = 0; + const [small, large] = a.size <= b.size ? [a, b] : [b, a]; + for (const x of small) if (large.has(x)) n++; + return n; +} + +/** a − b as a sorted array of keys. */ +export function difference(a, b) { + const out = []; + for (const x of a) if (!b.has(x)) out.push(x); + return out.sort(); +} + +/** + * Core CIS-vs-AIS scorer. `cis` / `ais` are Sets of canonical symbol keys. + * + * Empty-denominator semantics are EXPLICIT (not silently 0 or 1): + * - |CIS|=0 ⇒ precision = null (no predictions to be right/wrong about). + * - |AIS|=0 ⇒ recall = null (nothing to find — this scope has no truth). + * - F1 = null whenever precision or recall is null OR both are 0. + * A null metric is REPORTED as `n/a`, never averaged in as 0 — collapsing it to + * 0 would punish a mode for a scope that simply has no ground truth (the + * apples-to-oranges trap, R1). + */ +export function score(cis, ais) { + const tp = intersectionSize(cis, ais); + const precision = cis.size === 0 ? null : tp / cis.size; + const recall = ais.size === 0 ? null : tp / ais.size; + let f1 = null; + if (precision !== null && recall !== null && precision + recall > 0) { + f1 = (2 * precision * recall) / (precision + recall); + } + return { + tp, + cisSize: cis.size, + aisSize: ais.size, + precision, + recall, + f1, + fpis: difference(cis, ais), // CIS − AIS (noise / over-approx) + fnis: difference(ais, cis), // AIS − CIS (missed / under-approx) + fpisCount: cis.size - tp, + fnisCount: ais.size - tp, + // |CIS|/|AIS| size ratio (>1 over-approximates, <1 under). null if |AIS|=0. + cisAisRatio: ais.size === 0 ? null : cis.size / ais.size, + }; +} + +/** + * Cross-mode comparison of two CIS sets against a shared AIS (KTD9 set-diffs). + * Jaccard(callgraph_CIS, pdg_CIS) + directional set-diffs, each split into + * `true` (∩AIS — a real find the other mode missed) vs `noise` (−AIS — a false + * positive the other mode avoided). + */ +export function compareModes(callgraphCis, pdgCis, ais) { + const union = new Set([...callgraphCis, ...pdgCis]); + const inter = intersectionSize(callgraphCis, pdgCis); + const jaccard = union.size === 0 ? null : inter / union.size; + + const pdgOnly = difference(pdgCis, callgraphCis); + const callgraphOnly = difference(callgraphCis, pdgCis); + const splitByAis = (keys) => { + const trueFinds = keys.filter((k) => ais.has(k)).sort(); + const noise = keys.filter((k) => !ais.has(k)).sort(); + return { all: keys, true: trueFinds, noise }; + }; + return { + jaccard, + intersectionSize: inter, + unionSize: union.size, + pdgOnly: splitByAis(pdgOnly), + callgraphOnly: splitByAis(callgraphOnly), + }; +} + +/** + * Aggregate per-case scores for ONE (mode, scope) into a corpus row. Averaging + * follows KTD9 "per change, averaged over the corpus": a case with a null metric + * (e.g. |CIS|=0 precision) is EXCLUDED from that metric's mean (counted in + * `nMetric`), never folded in as 0. The macro-average is over the cases that + * actually have the metric defined; `nCases` records the stratum size for the + * underpowered-corpus floor (F3). + */ +export function aggregate(perCaseScores) { + const avg = (sel) => { + const xs = perCaseScores.map(sel).filter((v) => v !== null && v !== undefined); + if (xs.length === 0) return { mean: null, n: 0 }; + return { mean: xs.reduce((a, b) => a + b, 0) / xs.length, n: xs.length }; + }; + const p = avg((s) => s.precision); + const r = avg((s) => s.recall); + const f = avg((s) => s.f1); + const ratio = avg((s) => s.cisAisRatio); + return { + nCases: perCaseScores.length, + precision: p.mean, + nPrecision: p.n, + recall: r.mean, + nRecall: r.n, + f1: f.mean, + nF1: f.n, + cisAisRatio: ratio.mean, + // Summed FPIS/FNIS counts over the stratum (totals, not means) — the + // absolute over/under-approximation volume. + fpis: perCaseScores.reduce((a, s) => a + (s.fpisCount ?? 0), 0), + fnis: perCaseScores.reduce((a, s) => a + (s.fnisCount ?? 0), 0), + }; +} + +/** + * Partition a mode's reported CIS keys into per-scope sub-CIS, given the + * criterion's own symbol key. INTRA = the criterion symbol itself (the only + * symbol whose blocks/edges are intra-procedural); INTER = every OTHER reported + * symbol (callees / cross-function reach). `unresolved` shadow entries (id null, + * surfaced under a file) are kept in INTER — they are non-criterion reach the + * mode could not attribute to a named symbol, and dropping them would hide a + * recall fact (R9). MIXED scope unions both. + */ +export function partitionCisByScope(cisKeys, criterionKey) { + const intra = new Set(); + const inter = new Set(); + for (const k of cisKeys) { + if (k === criterionKey) intra.add(k); + else inter.add(k); + } + return { intra, inter, mixed: new Set([...intra, ...inter]) }; +} + +/** + * Build the scope-appropriate AIS key sets from a ground-truth record. + * - intra: the criterion symbol itself (intra_AIS lines collapse onto it). A + * case with a non-empty intra_AIS contributes {criterion}; an empty intra_AIS + * contributes ∅ (no intra truth → recall n/a, not 0). + * - inter: the distinct callee symbols named in inter_AIS. + * - mixed: the union. + * Keys are `@` with paths normalised to the criterion's path + * style (the fixture annotations and the analyzer both use repo-relative + * `src/...` paths, so no rewrite is needed — asserted by Step 0). + */ +export function aisByScope(gt) { + const critKey = symbolKey(gt.criterion.name, gt.criterion.filePath); + const intra = new Set(); + if (Array.isArray(gt.intra_AIS) && gt.intra_AIS.length > 0) intra.add(critKey); + const inter = toKeySet( + (gt.inter_AIS ?? []).map((e) => ({ symbol: e.symbol, filePath: e.filePath })), + ); + return { criterionKey: critKey, intra, inter, mixed: new Set([...intra, ...inter]) }; +} + +const tagSymbolKeys = (keys) => new Set([...keys].map((k) => `symbol:${k}`)); +const tagLineKeys = (keys) => new Set([...keys].map((k) => `statement:${k}`)); + +/** + * Current callgraph unified CIS: inter-symbol axis only. The seed/criterion + * symbol is filtered because `inter_AIS` is cross-function by construction. + */ +export function callgraphUnifiedCis(gt, symbolCisKeys) { + const { criterionKey } = aisByScope(gt); + const inter = new Set([...symbolCisKeys].filter((k) => k !== criterionKey)); + return { intraLine: new Set(), interSymbol: tagSymbolKeys(inter) }; +} + +/** + * Unified PDG CIS: statement axis from affectedStatements plus, once runtime + * mode:'pdg' composes interprocedural reach, symbol axis from byDepth. The + * criterion symbol is filtered because inter_AIS is cross-function by + * construction. Passing only lineCisKeys preserves the old intra-only shape for + * focused metric tests. + */ +export function pdgUnifiedCis(lineCisKeys, symbolCisKeys = new Set(), gt = null) { + const { criterionKey } = gt ? aisByScope(gt) : { criterionKey: null }; + const inter = criterionKey + ? new Set([...symbolCisKeys].filter((k) => k !== criterionKey)) + : new Set(symbolCisKeys); + return { intraLine: tagLineKeys(lineCisKeys), interSymbol: tagSymbolKeys(inter) }; +} + +/** Evaluation-only composed baseline: callgraph inter-symbol + PDG intra-line. */ +export function composeUnifiedCis(...parts) { + const intraLine = new Set(); + const interSymbol = new Set(); + for (const part of parts) { + for (const k of part.intraLine ?? []) intraLine.add(k); + for (const k of part.interSymbol ?? []) interSymbol.add(k); + } + return { intraLine, interSymbol }; +} + +/** Score one engine/candidate against unified AIS without blending axes. */ +export function scoreUnifiedAxes(cis, ais) { + return { + intraLine: score(cis.intraLine ?? new Set(), ais.intraLine ?? new Set()), + interSymbol: score(cis.interSymbol ?? new Set(), ais.interSymbol ?? new Set()), + }; +} + +export function aggregateUnifiedScores(perCaseScores) { + const intraLine = aggregate(perCaseScores.map((s) => s.intraLine)); + const interSymbol = aggregate(perCaseScores.map((s) => s.interSymbol)); + const definedRecalls = [intraLine.recall, interSymbol.recall].filter( + (v) => v !== null && v !== undefined, + ); + return { + intraLine, + interSymbol, + minRecall: definedRecalls.length ? Math.min(...definedRecalls) : null, + fpis: (intraLine.fpis ?? 0) + (interSymbol.fpis ?? 0), + fnis: (intraLine.fnis ?? 0) + (interSymbol.fnis ?? 0), + }; +} + +/** + * Order-independent annotation-set fingerprint (KTD10). Mirrors the + * bench/cfg/measure.mjs canonicalization TECHNIQUE (sort every collection, + * stringify deterministically, hash) — but is annotation-set-shaped and written + * here, NOT a literal import of `canonicalizeCfg`. Any unreviewed edit to a + * ground-truth.json (criterion, AIS membership, locus, direction, edge kinds) + * changes the digest, tripping a `--check` gate distinct from the F1 band. + * + * `hash` is injected (node:crypto in the harness; a stub in the unit test) so + * this module pulls no node-only deps that would complicate the test import. + */ +export function canonicalizeAnnotationSet(fixtures) { + const canonAis = (entries) => + (entries ?? []) + .map((e) => `${e.symbol}|${e.filePath}|${e.line ?? '-'}`) + .sort() + .join(';'); + // U9 resolved-id soundness block (optional): the expected id-proven set, the + // name-match over-attribution set, and the eliminated collision id(s). It is + // part of the ground truth — an unreviewed edit changes the gate, so it MUST + // trip the fingerprint. Sorted so the digest is order-independent; absent on + // fixtures without an `idBridge` block (canonicalized as `-`). + const canonIdBridge = (b) => { + const sorted = (xs) => [...(xs ?? [])].sort().join(','); + return b + ? `${b.seedLine ?? '-'}|${sorted(b.idProven)}|${sorted(b.nameWouldProve)}|${sorted(b.fpEliminated)}` + : '-'; + }; + const lines = fixtures + .map((fx) => { + const c = fx.gt.criterion; + const kinds = Array.isArray(c.pdgEdgeKinds) ? [...c.pdgEdgeKinds].sort().join(',') : '-'; + // `line` is the criterion's 1-based statement anchor (U7 — the seed of the + // statement-anchored PDG slice). It is part of the ground truth: changing + // which statement the slice seeds on changes the measured PDG impact set, + // so an unreviewed `criterion.line` edit MUST trip the fingerprint gate. + return [ + `case=${fx.name}`, + `schema=${fx.gt.schemaVersion}`, + `crit=${c.name}|${c.filePath}|${c.direction}|${c.line ?? '-'}|${c.marker ?? '-'}|${kinds}`, + `locus=${fx.gt.locus}`, + `pdgScoring=${fx.gt.pdgScoring ?? '-'}`, + `provenance=${fx.gt.provenance}`, + `intra=${canonAis(fx.gt.intra_AIS)}`, + `inter=${canonAis(fx.gt.inter_AIS)}`, + `idBridge=${canonIdBridge(fx.gt.idBridge)}`, + ].join('\n'); + }) + .sort() + .join('\n====\n'); + return lines; +} + +/** SHA-256 the canonical string with an injected hashing function. */ +export function fingerprintAnnotationSet(fixtures, sha256Hex) { + return sha256Hex(canonicalizeAnnotationSet(fixtures)); +} + +// ── U2: mutation/dynamic-oracle PURE scorers (dependency-free, unit-testable) ─ +// +// `behavioralAis` (B) is the dynamic forward slice the oracle PROVED by value-diff; +// `slice` is the SAME live static PDG slice the F1 metric scores (pdgLineCis of +// `affectedStatements`); `manualAis` (M) is the manual `intra_AIS` line set. All +// three are plain string Sets of `:` keys. These functions are +// pure math over those sets, so `test/unit/impact-pdg-metric-math.test.ts` asserts +// them deterministically (no DB / analyze / Babel / random). + +/** + * mutation_recall = |B ∩ slice| / |B|. A recall < 1.0 means B ∖ slice is a + * statement the dynamic oracle PROVED depends on the criterion that the static + * slice MISSED — a real recall hole (or a known U1 ascent gap). |B|=0 ⇒ recall + * is `null` (the oracle proved nothing to find — equivalent mutants only, or an + * oracle-excluded fixture), never 0. `missing` = B ∖ slice (the dangerous miss), + * `extra` = slice ∖ B (sound static over-approximation; reported, NOT gated). + */ +export function mutationRecall(behavioralAis, slice) { + const B = behavioralAis instanceof Set ? behavioralAis : new Set(behavioralAis); + const S = slice instanceof Set ? slice : new Set(slice); + const tp = intersectionSize(B, S); + const recall = B.size === 0 ? null : tp / B.size; + return { + recall, + bSize: B.size, + sliceSize: S.size, + intersection: tp, + missing: difference(B, S), // B ∖ slice — recall hole (sorted) + extra: difference(S, B), // slice ∖ B — sound over-approximation (informational) + }; +} + +/** + * Circularity cross-check: B ∖ M. A NON-EMPTY result means the manual annotation + * MISSED a real dependence the dynamic oracle proved — the headline independent + * evidence U2 exists to produce. Reported as a WARN with the specific lines; it + * is NOT a failure (the corpus documents annotation incompleteness as threat #1). + * `confirmed` = B ∩ M (the manual lines the oracle independently re-derived). + */ +export function circularityDiff(behavioralAis, manualAis) { + const B = behavioralAis instanceof Set ? behavioralAis : new Set(behavioralAis); + const M = manualAis instanceof Set ? manualAis : new Set(manualAis); + return { + beyondManual: difference(B, M), // B ∖ M — manual missed these (WARN) + confirmed: [...B].filter((k) => M.has(k)).sort(), // B ∩ M + manualOnly: difference(M, B), // M ∖ B — manual claimed, oracle did not prove + }; +} + +/** + * A mutant is EQUIVALENT iff its behavioral AIS is empty (it changed no observed + * value at any non-criterion line on any input). Equivalent mutants are discarded + * from the union (they carry no dependence signal). Accepts the oracle's per-mutant + * `{ diffLines }` record or a bare line array/Set. + */ +export function isEquivalentMutant(behavioralAis) { + const lines = Array.isArray(behavioralAis?.diffLines) ? behavioralAis.diffLines : behavioralAis; + const set = lines instanceof Set ? lines : new Set(lines ?? []); + return set.size === 0; +} + +/** + * Order-independent canonical string over the mutation-oracle output set (one + * entry per fixture: criterion key + sorted behavioral AIS + sorted non-equivalent + * mutant ops). Mirrors the annotation-fingerprint TECHNIQUE so a CHANGE in what the + * oracle proves is detectable. `hash` is injected (node:crypto in the harness; a + * stub in the unit test) so this module pulls no node-only deps. + */ +export function canonicalizeMutationSet(perFixture) { + return perFixture + .map((f) => { + const ais = [...(f.behavioralAis ?? [])].sort().join(','); + const ops = [...(f.mutants ?? [])] + .filter((m) => !isEquivalentMutant(m)) + .map((m) => m.op) + .sort() + .join(','); + return [`case=${f.name}`, `crit=${f.criterionKey ?? '-'}`, `ais=${ais}`, `ops=${ops}`].join( + '\n', + ); + }) + .sort() + .join('\n====\n'); +} + +export function fingerprintMutationSet(perFixture, sha256Hex) { + return sha256Hex(canonicalizeMutationSet(perFixture)); +} + +/** median of a numeric array (substrate-stability gate, F5). */ +export function median(xs) { + if (xs.length === 0) return null; + const s = [...xs].sort((a, b) => a - b); + const m = Math.floor(s.length / 2); + return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2; +} diff --git a/gitnexus/bench/impact-pdg/mutation-oracle.mjs b/gitnexus/bench/impact-pdg/mutation-oracle.mjs new file mode 100644 index 000000000..72774a573 --- /dev/null +++ b/gitnexus/bench/impact-pdg/mutation-oracle.mjs @@ -0,0 +1,541 @@ +/** + * U2 — SOUND mutation/dynamic ground-truth ORACLE for the impact-PDG forward + * slice. An INDEPENDENT check on the hand-annotated `intra_AIS`: the PR's #1 + * declared validity threat is annotation circularity, so this module derives a + * REAL DYNAMIC FORWARD SLICE by VALUE-DIFF (Infection + Propagation — NOT mere + * coverage), and the harness cross-checks it against both the static PDG slice + * (recall) and the manual annotation (circularity). + * + * Research grounding (the design this implements): + * - Agrawal & Horgan, "Dynamic Program Slicing", PLDI'90 — a dynamic slice is + * the set of statements that actually affected the criterion on an execution. + * - Tip, "A Survey of Program Slicing Techniques" (1995) — static ⊇ dynamic for + * a sound static slicer on the executed paths. + * - Voas, "PIE / propagation-infection-execution", TSE'92 — a fault is observed + * only when it is executed (E), infects state (I), and PROPAGATES (P) to an + * observable point. Coverage alone is only E; dependence needs I+P = an actual + * VALUE CHANGE. So `behavioral_AIS` is computed from value diffs, not coverage. + * + * ── What it does, per fixture ─────────────────────────────────────────────── + * 1. MUTATE the criterion line ONLY (≤4 mutants, line-scoped regex operators: + * AOR, ROR, LCR, CRP, UOI). Discard EQUIVALENT mutants (empty behavioral_AIS). + * 2. Derive inputs via a tiny TYPE-DRIVEN generator from the criterion fn's + * params (number, number[], boolean, string). Multi-input covers both arms. + * 3. INSTRUMENT the ORIGINAL TS AST with a value-transparent `__trace` wrapper on + * VariableDeclarator.init / AssignmentExpression RHS / ReturnStatement.arg / + * CallExpression — recording `(filePath:line, occ) -> serialized value` and + * returning the expression unchanged. loc lines are 1-based filePath:line in + * the SAME space as the static slice (no source-map needed). + * 4. behavioral_AIS = { filePath:line where serialize(orig) != serialize(mut) + * for some input/occurrence }, EXCLUDING the criterion line, unioned over + * inputs then over non-equivalent mutants. + * + * NO production/src import. Pure ESM + Babel + tsx dynamic-import. All generated/ + * instrumented artifacts live under an os.tmpdir() dir (gn-impact-pdg-mut-*), + * NEVER inside fixtures/. The only persisted file is the per-fixture + * `mutation-ground-truth.json` SIDECAR (data, separate from the manual + * `ground-truth.json`, never overwriting it). + */ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { pathToFileURL } from 'node:url'; +import { parse } from '@babel/parser'; +import _traverse from '@babel/traverse'; +import _generate from '@babel/generator'; +import * as t from '@babel/types'; + +const traverse = _traverse.default ?? _traverse; +const generate = _generate.default ?? _generate; + +// ── deterministic value serializer ─────────────────────────────────────────── +// undefined -> '#u', NaN/Infinity handled, stable key order for plain objects. +// A value change / appearance / disappearance is what makes a line "behavioral". +export function serializeValue(v) { + if (v === undefined) return '#u'; + if (v === null) return '#n'; + if (typeof v === 'number') { + if (Number.isNaN(v)) return '#NaN'; + if (v === Infinity) return '#+Inf'; + if (v === -Infinity) return '#-Inf'; + return 'n:' + String(v); + } + if (typeof v === 'boolean') return 'b:' + (v ? '1' : '0'); + if (typeof v === 'string') return 's:' + v; + if (typeof v === 'bigint') return 'B:' + v.toString(); + if (typeof v === 'function') return 'fn'; + if (Array.isArray(v)) return '[' + v.map(serializeValue).join(',') + ']'; + if (typeof v === 'object') { + const keys = Object.keys(v).sort(); + return '{' + keys.map((k) => k + '=' + serializeValue(v[k])).join(',') + '}'; + } + return String(v); +} + +// ── type-driven input generator ─────────────────────────────────────────────── +// number -> [5, -3, 0]; number[] -> [[1,2,3], [-1,-2], []]; boolean -> [true,false]; +// string -> ['a','b','z']. Multi-input is REQUIRED to cover both branch arms. +const TYPE_VALUES = { + number: [5, -3, 0], + 'number[]': [[1, 2, 3], [-1, -2], []], + boolean: [true, false], + string: ['a', 'b', 'z'], + unknown: [0], +}; + +function normalizeTypeAnnotation(node) { + if (!node) return 'unknown'; + // node is a TSTypeAnnotation wrapper; unwrap to the inner type. + const ty = node.typeAnnotation ?? node; + if (t.isTSNumberKeyword(ty)) return 'number'; + if (t.isTSBooleanKeyword(ty)) return 'boolean'; + if (t.isTSStringKeyword(ty)) return 'string'; + if (t.isTSArrayType(ty)) { + if (t.isTSNumberKeyword(ty.elementType)) return 'number[]'; + return 'unknown'; + } + return 'unknown'; +} + +/** + * Cartesian product of per-parameter candidate value lists, capped so the run + * stays cheap. Returns an array of argument tuples. + */ +function inputTuplesFor(paramTypes, cap = 6) { + let tuples = [[]]; + for (const ty of paramTypes) { + const vals = TYPE_VALUES[ty] ?? TYPE_VALUES.unknown; + const next = []; + for (const partial of tuples) { + for (const v of vals) { + next.push([...partial, v]); + if (next.length >= cap * 4) break; + } + } + tuples = next; + } + // Deduplicate by serialized tuple, then cap. + const seen = new Set(); + const out = []; + for (const tup of tuples) { + const key = tup.map(serializeValue).join('|'); + if (seen.has(key)) continue; + seen.add(key); + out.push(tup); + if (out.length >= cap) break; + } + return out; +} + +// ── line-scoped mutation operators (regex on the criterion line text) ───────── +// AOR arithmetic, ROR relational, LCR logical, CRP numeric-literal, UOI negate. +// Each yields ≤1 replacement per applicable token; we cap the total at 4. +function lineMutants(rawLine) { + // Split off a trailing line-comment so operators never mutate inside it (a `-` + // injected into a comment string is harmless but a non-greedy UOI wrap would + // otherwise swallow the comment and produce invalid syntax). The comment is + // re-appended verbatim to every mutant so loc lines are preserved. + const cm = rawLine.match(/^(.*?)(\s*\/\/.*)$/); + const line = cm ? cm[1] : rawLine; + const comment = cm ? cm[2] : ''; + const mutants = []; + const push = (text, op) => { + const full = text + comment; + if (full !== rawLine && !mutants.some((m) => m.text === full)) mutants.push({ text: full, op }); + }; + + // AOR: swap the FIRST binary arithmetic operator. Order matters — try + then - + // etc. so a line with `a + 1` mutates `+`. + const aor = [ + [/(?<=[\w)\]\s])\+(?=[\s\w(])/, '-'], + [/(?<=[\w)\]\s])-(?=[\s\w(])/, '+'], + [/(?<=[\w)\]\s])\*(?=[\s\w(])/, '/'], + [/(?<=[\w)\]\s])\/(?=[\s\w(])/, '*'], + [/(?<=[\w)\]\s])%(?=[\s\w(])/, '*'], + ]; + for (const [re, rep] of aor) { + if (re.test(line)) { + push(line.replace(re, rep), 'AOR'); + break; + } + } + + // ROR: relational/equality. Longer operators first so `<=` is not split. + const ror = [ + [/===/, '!=='], + [/!==/, '==='], + [/<=/, '>'], + [/>=/, '<'], + [/(?=!])<(?![<=])/, '>='], + [/(?=!])>(?![>=])/, '<='], + ]; + for (const [re, rep] of ror) { + if (re.test(line)) { + push(line.replace(re, rep), 'ROR'); + break; + } + } + + // LCR: logical connector / negation. Connectors first; then unary `!` flip on a + // guard predicate (`if (!ok)` ⇒ `if (ok)`, a real control-flow change). + const lcr = [ + [/\|\|/, '&&'], + [/&&/, '||'], + ]; + let lcrApplied = false; + for (const [re, rep] of lcr) { + if (re.test(line)) { + push(line.replace(re, rep), 'LCR'); + lcrApplied = true; + break; + } + } + if (!lcrApplied) { + const neg = line.match(/(?<=[(\s])!(?=[\w(])/); + if (neg) push(line.replace(/(?<=[(\s])!(?=[\w(])/, ''), 'LCR'); + } + + // CRP: first standalone numeric literal -> k+1 (and 0 if not already 0). + const numMatch = line.match(/(? { + const node = nodePath.node; + if (!node || !node.loc) return; + // never re-wrap our own trace call + if (t.isCallExpression(node) && t.isIdentifier(node.callee, { name: '__trace' })) return; + const line = node.loc.start.line; + const o = (occ.get(line) ?? 0) + 1; + occ.set(line, o); + nodePath.replaceWith( + t.callExpression(t.identifier('__trace'), [ + node, + t.numericLiteral(line), + t.stringLiteral(filePath), + t.numericLiteral(o), + ]), + ); + nodePath.skip(); + }; + traverse(ast, { + VariableDeclarator(p) { + if (p.node.init) wrap(p.get('init')); + }, + AssignmentExpression(p) { + // wrap the RHS; the assignment value itself is observed at its own line via + // the declarator/return sites, so wrapping the RHS captures the new value. + if (p.node.right) wrap(p.get('right')); + }, + ReturnStatement(p) { + if (p.node.argument) wrap(p.get('argument')); + }, + CallExpression(p) { + // a bare call statement (effectful) — wrap so its return value/occurrence is + // observed. Skips our own __trace / __tick instrumentation calls. + if ( + t.isIdentifier(p.node.callee, { name: '__trace' }) || + t.isIdentifier(p.node.callee, { name: '__tick' }) + ) + return; + wrap(p); + }, + }); + // Bound every loop with a back-edge step budget: prepend `__tick()` to each loop + // body so a NON-TERMINATING mutant (e.g. a flipped operator that makes a loop + // never exit) throws `__GN_NONTERM` instead of hanging the in-process run. + // Recursion self-terminates via stack overflow, so only loops need this guard. + traverse(ast, { + 'ForStatement|ForInStatement|ForOfStatement|WhileStatement|DoWhileStatement'(p) { + p.ensureBlock(); + p.get('body').unshiftContainer( + 'body', + t.expressionStatement(t.callExpression(t.identifier('__tick'), [])), + ); + }, + }); + const body = generate(ast, { retainLines: true }).code; + const preamble = + 'const __traceLog=[];\n' + + 'let __ticks=0;\n' + + 'function __tick(){ if(++__ticks>50000){ throw new Error("__GN_NONTERM"); } }\n' + + 'function __trace(v,line,file,occ){__traceLog.push({line,file,occ,v});return v;}\n' + + 'export {__traceLog as __GN_TRACE_LOG};\n'; + return preamble + body; +} + +// ── run one instrumented module on a tuple, collect (line:occ -> serialized) ── +async function runTraced(moduleFile, fnName, args) { + // bust the import cache so the original and each mutant are distinct modules. + const url = pathToFileURL(moduleFile).href + `?v=${Math.random().toString(36).slice(2)}`; + const mod = await import(url); + const log = mod.__GN_TRACE_LOG; + log.length = 0; + const fn = mod[fnName]; + if (typeof fn !== 'function') { + throw new Error(`instrumented module has no exported function ${fnName}`); + } + let threw = null; + let nonTerminating = false; + try { + fn(...args); + } catch (e) { + const msg = e instanceof Error ? e.message : String(e); + if (msg.includes('__GN_NONTERM')) nonTerminating = true; + threw = msg; + } + // A non-terminating mutant yields no finite value trace — return empty so the + // diff SKIPS it (logged, excluded from behavioral_AIS) rather than letting its + // partial loop trace fabricate spurious per-iteration diffs. + if (nonTerminating) return { observed: new Map(), threw, nonTerminating: true }; + // (filePath:line, occ) -> serialized value. A line may appear multiple times; + // key on occurrence so a per-iteration value change is observed. + const observed = new Map(); + for (const rec of log) { + observed.set(`${rec.file}:${rec.line}#${rec.occ}`, serializeValue(rec.v)); + } + return { observed, threw, nonTerminating: false }; +} + +/** parse `function name(params)` signatures to get the criterion's param types. */ +function criterionParamTypes(src, fnName) { + const ast = parse(src, { sourceType: 'module', plugins: ['typescript'] }); + let types = []; + traverse(ast, { + 'FunctionDeclaration|FunctionExpression|ArrowFunctionExpression'(p) { + const id = p.node.id; + const isMatch = + (id && id.name === fnName) || + (t.isVariableDeclarator(p.parent) && + t.isIdentifier(p.parent.id) && + p.parent.id.name === fnName); + if (!isMatch) return; + types = p.node.params.map((param) => { + const ann = t.isIdentifier(param) ? param.typeAnnotation : param.typeAnnotation; + return normalizeTypeAnnotation(ann); + }); + p.stop(); + }, + }); + return types; +} + +/** + * Derive the behavioral (dynamic) AIS for a fixture. + * + * @param {{name:string, dir:string, gt:object}} fx — fixture record (gt is the + * manual ground-truth.json). + * @param {string} workDir — the SAME temp working copy the analyze step used + * (so `workDir/src/` line numbers align with the static slice keys). + * @returns {Promise<{ + * behavioralAis: string[], // sorted `:` keys (criterion excluded) + * criterionLine: number, + * criterionKey: string, + * filePath: string, + * inputs: unknown[][], // the tuples used + * paramTypes: string[], + * mutants: {op:string, text:string, equivalent:boolean, diffLines:string[]}[], + * skipped: null|string, // a reason if the oracle could not run + * }>} + */ +export async function deriveBehavioralAis(fx, workDir) { + const filePath = fx.gt.criterion.filePath; // repo-relative `src/...` + const fnName = fx.gt.criterion.name; + const criterionLine = fx.gt.criterion.line; + const criterionKey = `${filePath}:${criterionLine}`; + const absSrc = path.join(workDir, filePath); + + const empty = { + behavioralAis: [], + criterionLine: criterionLine ?? null, + criterionKey, + filePath, + inputs: [], + paramTypes: [], + mutants: [], + skipped: null, + }; + + if (!criterionLine || !fs.existsSync(absSrc)) { + return { ...empty, skipped: `criterion file/line missing (${absSrc}:${criterionLine})` }; + } + + const src = fs.readFileSync(absSrc, 'utf8'); + const srcLines = src.split('\n'); + const lineText = srcLines[criterionLine - 1]; + if (lineText === undefined) { + return { ...empty, skipped: `criterion line ${criterionLine} out of range` }; + } + + const paramTypes = criterionParamTypes(src, fnName); + const inputs = inputTuplesFor(paramTypes); + if (inputs.length === 0) { + return { ...empty, paramTypes, skipped: 'no inputs derivable' }; + } + + const mutantSpecs = lineMutants(lineText); + if (mutantSpecs.length === 0) { + return { ...empty, paramTypes, inputs, skipped: 'no applicable mutation operator on line' }; + } + + const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-impact-pdg-mut-')); + const mutants = []; + const aisLines = new Set(); + try { + // Instrument the ORIGINAL once; copy the whole src tree so cross-file imports + // (if any) still resolve, then overwrite the criterion file. + fs.cpSync(path.join(workDir, 'src'), path.join(tmp, 'src'), { recursive: true }); + const instrumentedOriginal = instrument(src, filePath); + const origFile = path.join(tmp, filePath); + fs.writeFileSync(origFile, instrumentedOriginal); + + // Baseline traces (one per input tuple). + const baseline = []; + for (const args of inputs) baseline.push(await runTraced(origFile, fnName, args)); + + let mi = 0; + for (const spec of mutantSpecs) { + mi += 1; + const mutLines = srcLines.slice(); + mutLines[criterionLine - 1] = spec.text; + const mutSrc = mutLines.join('\n'); + // Instrument the MUTANT source (same loc space — the mutated line keeps its + // line number; retainLines preserves all other lines). A regex operator can + // occasionally produce syntactically invalid TS (e.g. a `-` injected where a + // unary context makes it ambiguous); such a mutant is not a valid program, so + // it is SKIPPED (recorded as invalid, never crashes the pass). + let instrumentedMutant; + try { + instrumentedMutant = instrument(mutSrc, filePath); + } catch { + mutants.push({ + op: spec.op, + text: spec.text, + equivalent: true, + invalid: true, + diffLines: [], + }); + continue; + } + const mutFile = path.join(tmp, `mut${mi}__${path.basename(filePath)}`); + fs.writeFileSync(mutFile, instrumentedMutant); + + const diffLines = new Set(); + let nonTerminating = false; + for (let i = 0; i < inputs.length; i++) { + const mutRun = await runTraced(mutFile, fnName, inputs[i]); + if (mutRun.nonTerminating) { + nonTerminating = true; + break; + } + const base = baseline[i]; + // union of keys observed in either run (a value can appear/disappear). + const keys = new Set([...base.observed.keys(), ...mutRun.observed.keys()]); + for (const k of keys) { + const bv = base.observed.get(k); + const mv = mutRun.observed.get(k); + if (bv !== mv) { + // strip the occurrence suffix back to a `:` key. + const lineOnly = k.slice(0, k.lastIndexOf('#')); + diffLines.add(lineOnly); + } + } + // A divergence in throw-behaviour is itself propagation to the return + // point: attribute it to the criterion line's continuation. We DON'T add + // the criterion line (excluded below), but a differing throw with no value + // diff still implies the function-result line changed — captured via the + // return-line value diff already (return not reached => key disappears). + } + if (nonTerminating) { + // The criterion mutation made a downstream loop non-terminating. This proves + // divergence but yields no observable per-statement value trace, so it is + // LOGGED and EXCLUDED from behavioral_AIS — excluding it keeps recall sound + // (behavioral_AIS stays a subset of the true dynamic forward slice). + mutants.push({ + op: spec.op, + text: spec.text, + equivalent: false, + nonTerminating: true, + diffLines: [], + }); + continue; + } + // EXCLUDE the criterion line itself. + diffLines.delete(criterionKey); + const diffArr = [...diffLines].sort(); + const equivalent = diffArr.length === 0; + mutants.push({ op: spec.op, text: spec.text, equivalent, diffLines: diffArr }); + // EQUIVALENT mutants (empty behavioral set) are DISCARDED from the union. + if (!equivalent) for (const l of diffArr) aisLines.add(l); + } + } finally { + fs.rmSync(tmp, { recursive: true, force: true }); + } + + return { + behavioralAis: [...aisLines].sort(), + criterionLine, + criterionKey, + filePath, + inputs, + paramTypes, + mutants, + skipped: null, + }; +} + +/** + * Write/refresh the per-fixture audit sidecar (provenance:'mutation'). SEPARATE + * from the manual ground-truth.json — never overwrites it. Returns the path. + */ +export function writeMutationSidecar(fx, derived) { + const out = { + schemaVersion: 1, + provenance: 'mutation', + criterion: { + name: fx.gt.criterion.name, + filePath: derived.filePath, + line: derived.criterionLine, + }, + paramTypes: derived.paramTypes, + inputs: derived.inputs, + behavioral_AIS: derived.behavioralAis, + mutants: derived.mutants, + skipped: derived.skipped, + note: + 'AUTO-GENERATED dynamic forward-slice oracle (U2). VALUE-DIFF behavioral AIS ' + + '(Infection+Propagation, not coverage). Regenerated by `measure.mjs --mutation`. ' + + 'Independent cross-check of the manual ground-truth.json — NOT a hand annotation.', + }; + const sidecarPath = path.join(fx.dir, 'mutation-ground-truth.json'); + fs.writeFileSync(sidecarPath, JSON.stringify(out, null, 2) + '\n'); + return sidecarPath; +} diff --git a/gitnexus/bench/impact-pdg/name-collision.mjs b/gitnexus/bench/impact-pdg/name-collision.mjs new file mode 100644 index 000000000..9a62e86eb --- /dev/null +++ b/gitnexus/bench/impact-pdg/name-collision.mjs @@ -0,0 +1,599 @@ +/** + * Realized name-collision probe for the PDG-impact statement-precise bridge. + * + * The bridge labels a callgraph-reached callee "proven" (callgraph-bridge) iff its + * LEAF NAME appears in the changed line's dependence-slice block callees + * (`pdgBridgeEvidenceForImpact`, pdg-impact.ts). Because the match is by name, two + * distinct reached symbols that share a leaf name (e.g. two `get`s) are BOTH proven + * whenever that name is in the slice — but the slice's call site(s) resolve to a + * specific subset, so the extras are over-attribution (false-proven). This is the + * documented "conservative SUPERSET" caveat (pdg-impact.ts:1352). + * + * This probe QUANTIFIES that over-attribution on real code, to decide whether a + * sound resolved-symbol-id bridge is worth building. + * + * Key property that makes an index-only measurement rigorous: a collision + * false-positive is ALWAYS a name-ambiguous proven label. To be proven a symbol + * must first be reached, so every same-name over-attribution surfaces as >=2 + * DISTINCT proven symbol-ids sharing one leaf name. Counting those is therefore + * COMPLETE for collision-FP. It is an UPPER BOUND (a slice could legitimately call + * two same-named callees on different lines, in which case both proven labels are + * correct), so the realized FP is in [0, ambiguous]. A near-zero result is a + * decisive no-go; a material result motivates the exact line-join confirmation + * (which needs a re-run that captures per-call-site resolved ids — the persisted + * CALLS edge has no call-site line, BasicBlock.callees is a deduped leaf-name set). + * + * Focus is DEPTH 1: name-matching only fires at the first hop; deeper proven labels + * are inherited from their depth-1 ancestor (betterBridgeEvidence), a different + * (transitive) imprecision, not name collision. + * + * ── U8: realized id-vs-name diff (the R5 proof, exact-slice) ───────────────── + * After U5/U6 the live bridge matches RESOLVED callee symbol-ids: on an index that + * carries `BasicBlock.calleeIds`, `pdgInterprocedural.statementPreciseByDepth` is + * the SOUND id-proven set. The `ambiguityRate` above then collapses to ~0 (ids + * discriminate same-named callees) — itself proof the collision is gone. To + * MEASURE what the id bridge changed versus the old name bridge on the SAME real + * slices, this probe also recomputes, per function, the NAME-proven set the + * leaf-name predicate WOULD prove on the EXACT seed∪reachable slice (not a + * function-level proxy — the impact result exposes `seedBlocks`/`reachableBlocks`, + * so we query those exact blocks' `callees` and replicate the bridge's + * `sliceCalleeNames.has(name)` fallback), and diffs the two reached-item sets: + * - fpEliminated = name-proven ∖ id-proven (collision FALSE-POSITIVES removed: + * labels the name match would prove that the id match drops) + * - fnRecovered = id-proven ∖ name-proven (import-alias FALSE-NEGATIVES + * recovered: labels the id match proves that the name match would miss) + * Both sets are keyed by reached-item id (resolved symbol-id), falling back to + * `name@filePath` for the rare id-less reached item. The scoring is factored into + * the dependency-free pure `scoreIdVsName`/`summarizeIdVsName` (asserted by + * test/unit/impact-pdg-id-vs-name-metrics.test.ts), exactly like + * `summarizeBlastRadius` in blast-radius.mjs. + */ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { median, parseMarkdownRows } from './blast-radius.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const REPO_ROOT = path.resolve(__dirname, '..', '..'); + +// `CALLEES_TRUNCATED_SENTINEL` (cfg/emit.ts) — a slice block that hit the +// per-statement site cap marks its callee set INCOMPLETE; the bridge keeps such +// reach callgraph-equal (callee-unknown) rather than under-proving. +const CALLEES_TRUNCATED_SENTINEL = '*'; + +function round(value, digits = 3) { + if (value === null || value === undefined || Number.isNaN(value)) return null; + const scale = 10 ** digits; + return Math.round(value * scale) / scale; +} + +function readOption(argv, name, fallback = undefined) { + const eq = argv.find((arg) => arg.startsWith(`--${name}=`)); + if (eq) return eq.slice(name.length + 3); + const idx = argv.indexOf(`--${name}`); + if (idx >= 0 && idx + 1 < argv.length) return argv[idx + 1]; + return fallback; +} + +function hasFlag(argv, name) { + return argv.includes(`--${name}`); +} + +function fmt(value, digits = 2) { + return value === null || value === undefined ? 'n/a' : Number(value).toFixed(digits); +} + +/** + * Proven (callgraph-bridge) items at a given depth from a `statementPreciseByDepth` + * record. Items carry { id, name, ... }; the projection already dropped + * unproven-bridge, so every item here is a proven label. + */ +function provenItemsAtDepth(byDepth, depth) { + const items = byDepth?.[depth] ?? byDepth?.[String(depth)] ?? []; + return Array.isArray(items) ? items : []; +} + +/** Full depth-1 inter-procedural reach (proven + unproven) — the direct callees. */ +function provGetReachedD1(pdg) { + const byDepth = pdg?.interproceduralByDepth ?? pdg?.pdgInterprocedural?.byDepth ?? {}; + return provenItemsAtDepth(byDepth, 1); +} + +/** + * Group proven items by leaf name, counting DISTINCT symbol ids per name. A name + * mapping to >=2 distinct ids is an ambiguous (non-discriminating) proven group: + * the name-match proved all of them, but the slice resolves to a subset. + */ +function nameCollisionStats(provenItems) { + const idsByName = new Map(); + for (const it of provenItems) { + if (!it || typeof it !== 'object') continue; + const name = typeof it.name === 'string' ? it.name : ''; + if (!name) continue; + const id = + typeof it.id === 'string' && it.id + ? it.id + : `${name}@${typeof it.filePath === 'string' ? it.filePath : '?'}`; + let set = idsByName.get(name); + if (!set) { + set = new Set(); + idsByName.set(name, set); + } + set.add(id); + } + let provenLabels = 0; + let ambiguousLabels = 0; // proven labels whose name is shared by >=2 distinct ids + let excessLabels = 0; // sum(count - 1) over ambiguous names = central FP estimate + const ambiguousNames = []; + for (const [name, ids] of idsByName) { + const c = ids.size; + provenLabels += c; + if (c >= 2) { + ambiguousLabels += c; + excessLabels += c - 1; + ambiguousNames.push({ name, count: c }); + } + } + ambiguousNames.sort((a, b) => b.count - a.count); + return { provenLabels, ambiguousLabels, excessLabels, ambiguousNames }; +} + +export function summarize(cases) { + const withProven = cases.filter((c) => c.provenLabels > 0); + const sum = (sel) => cases.reduce((a, c) => a + sel(c), 0); + const totalProven = sum((c) => c.provenLabels); + const totalAmbiguous = sum((c) => c.ambiguousLabels); + const totalExcess = sum((c) => c.excessLabels); + const totalReachedD1 = sum((c) => c.reachedD1); + const totalDivergent = sum((c) => c.divergentReached); + return { + n: cases.length, + functionsWithProvenLabels: withProven.length, + functionsWithAmbiguity: cases.filter((c) => c.ambiguousLabels > 0).length, + totalProvenLabels: totalProven, + totalAmbiguousLabels: totalAmbiguous, + totalExcessLabels: totalExcess, + // Upper bound on collision-FP as a fraction of all proven labels. + ambiguityRate: totalProven > 0 ? round(totalAmbiguous / totalProven) : null, + // Central FP estimate (assumes ~1 distinct resolved symbol per slice leaf name). + excessRate: totalProven > 0 ? round(totalExcess / totalProven) : null, + medianProvenPerFn: median(withProven.map((c) => c.provenLabels)), + // FN / aliasing axis: depth-1 reached callees whose resolved name is in NO + // block leaf of the owning function (alias/rename/dynamic) — the surface where + // a truly-on-slice callee can never be name-proven. + totalReachedD1, + totalDivergentReached: totalDivergent, + divergenceRate: totalReachedD1 > 0 ? round(totalDivergent / totalReachedD1) : null, + functionsWithDivergence: cases.filter((c) => c.divergentReached > 0).length, + }; +} + +/** + * PURE replica of the bridge's depth-1 evidence predicate + * (`pdgBridgeEvidenceForImpact`, pdg-impact.ts) over the SAME inputs, returning + * the two counterfactual proven sets for ONE function's slice: + * - `nameProven` = what the LEAF-NAME bridge would prove, + * - `idProven` = what the RESOLVED-ID bridge proves. + * + * The predicate (mirrored exactly so the whole-symbol / sentinel fallbacks cancel + * on both sides and ONLY the discriminating divergence survives the diff): + * 1. sliceCalleeNames empty → prove ALL (whole-symbol fallback) + * 2. sentinel ('*') in sliceCalleeNames → prove ALL (callee-unknown, capped) + * 3a. id path (ids present): prove iff item.id ∈ sliceCalleeIds + * 3b. name path (no ids): prove iff item.name ∈ sliceCalleeNames + * The name-counterfactual ALWAYS uses 3b at step 3 (what name-match would decide); + * the id set uses 3a when ids are present, else 3b (graceful degrade — identical + * to the name set on a pre-v3 index, so the diff is then structurally empty). + * + * `discriminating` flags the only regime where the two can differ (names present, + * no sentinel, ids present); non-discriminating slices fall back identically and + * contribute 0 to fpEliminated/fnRecovered by construction. + * + * Pure: no DB / analyze / Date / random — deterministic over plain inputs. + */ +export function bridgeProvenSets(reachedItems, sliceCalleeNames, sliceCalleeIds) { + const names = + sliceCalleeNames instanceof Set ? sliceCalleeNames : new Set(sliceCalleeNames ?? []); + const ids = sliceCalleeIds instanceof Set ? sliceCalleeIds : new Set(sliceCalleeIds ?? []); + const items = Array.isArray(reachedItems) ? reachedItems : []; + const wholeSymbol = names.size === 0; + const truncated = names.has(CALLEES_TRUNCATED_SENTINEL); + const idsPresent = ids.size > 0; + const discriminating = !wholeSymbol && !truncated && idsPresent; + + // Step 1/2: whole-symbol or sentinel ⇒ both bridges prove ALL reached items. + // Otherwise apply the step-3 membership predicate. + const fallbackProvesAll = wholeSymbol || truncated; + const provenBy = (member) => (fallbackProvesAll ? [...items] : items.filter(member)); + + const nameProven = provenBy((it) => typeof it?.name === 'string' && names.has(it.name)); + const idProven = provenBy((it) => + discriminating + ? typeof it?.id === 'string' && ids.has(it.id) + : typeof it?.name === 'string' && names.has(it.name), + ); + return { nameProven, idProven, discriminating, wholeSymbol, truncated }; +} + +/** + * Stable identity key for a reached/proven item. The resolved symbol id is the + * sound key (an alias/collision shares the leaf NAME but NEVER the id); fall back + * to `name@filePath` only for the rare id-less reached item (dynamic/unresolved), + * mirroring `symbolSetFromByDepth` in blast-radius.mjs. + */ +export function reachedItemKey(item) { + if (!item || typeof item !== 'object') return ''; + if (typeof item.id === 'string' && item.id.length > 0) return item.id; + const name = typeof item.name === 'string' ? item.name : '(unknown)'; + const filePath = typeof item.filePath === 'string' ? item.filePath : '(unknown)'; + return `${name}@${filePath}`; +} + +/** A de-duplicated key set over reached items (drops empty/unkeyable items). */ +function keySetOf(items) { + const out = new Set(); + for (const it of Array.isArray(items) ? items : []) { + const key = reachedItemKey(it); + if (key) out.add(key); + } + return out; +} + +/** + * PURE scorer (R5): diff the NAME-proven and ID-proven statement-precise sets for + * ONE function's slice. No DB / analyze / Date / random — deterministic over plain + * reached-item arrays so the unit test can assert the arithmetic. + * + * - `fpEliminated` = |name-proven ∖ id-proven| — collision false-positives the + * id bridge removed (the name predicate would prove these; the id predicate + * drops them because their resolved id is not on the exact slice). + * - `fnRecovered` = |id-proven ∖ name-proven| — import-alias false-negatives the + * id bridge recovered (the id predicate proves these via the resolved id even + * though their leaf name is absent from the slice's `callees`). + * + * Identical sets ⇒ both counts 0. Keys are sorted so the output is order-stable + * regardless of input ordering (determinism). + */ +export function scoreIdVsName(nameProvenItems, idProvenItems) { + const nameKeys = keySetOf(nameProvenItems); + const idKeys = keySetOf(idProvenItems); + const fpEliminatedKeys = [...nameKeys].filter((k) => !idKeys.has(k)).sort(); + const fnRecoveredKeys = [...idKeys].filter((k) => !nameKeys.has(k)).sort(); + return { + nameProven: nameKeys.size, + idProven: idKeys.size, + fpEliminated: fpEliminatedKeys.length, + fnRecovered: fnRecoveredKeys.length, + fpEliminatedKeys, + fnRecoveredKeys, + }; +} + +/** + * PURE aggregation over per-function `scoreIdVsName` outputs (the U8 headline). + * Dependency-free for the deterministic unit test, mirroring `summarizeBlastRadius`. + */ +export function summarizeIdVsName(cases) { + const sum = (sel) => cases.reduce((a, c) => a + sel(c), 0); + const totalNameProven = sum((c) => c.nameProven ?? 0); + const totalIdProven = sum((c) => c.idProven ?? 0); + const totalFpEliminated = sum((c) => c.fpEliminated ?? 0); + const totalFnRecovered = sum((c) => c.fnRecovered ?? 0); + return { + n: cases.length, + totalNameProven, + totalIdProven, + totalFpEliminated, + totalFnRecovered, + functionsWithFpEliminated: cases.filter((c) => (c.fpEliminated ?? 0) > 0).length, + functionsWithFnRecovered: cases.filter((c) => (c.fnRecovered ?? 0) > 0).length, + functionsWithDiscriminatingSlice: cases.filter((c) => c.discriminatingSlice === true).length, + // Fraction of name-proven labels the id bridge proved were over-attribution. + fpEliminatedRate: totalNameProven > 0 ? round(totalFpEliminated / totalNameProven) : null, + // Fraction of id-proven labels the name bridge would have missed (alias FN). + fnRecoveredRate: totalIdProven > 0 ? round(totalFnRecovered / totalIdProven) : null, + }; +} + +async function cypherRows(backend, repo, query) { + const res = await backend.callTool('cypher', { repo, query }); + return parseMarkdownRows(res?.markdown); +} + +/** + * Callee NAME set AND resolved-ID set for the EXACT dependence slice (seed ∪ + * reachable blocks the impact result exposes) — exactly the `sliceCalleeNames` / + * `sliceCalleeIds` the bridge unions over the slice (`local-backend.ts`). The + * sentinel `'*'` (a capped block) is PRESERVED in the name set so the replica + * predicate can take the callee-unknown fallback. This is the EXACT slice, NOT a + * function-level proxy. Block ids carry no quotes, so the `IN [...]` literal is + * safe (same convention as the candidate queries above). On a pre-v3 index the + * `calleeIds` column is absent → the query errors → ids degrade to empty (the + * bridge then name-matches, and the id-vs-name diff is structurally 0). + */ +async function sliceCalleeSetsOf(backend, repo, sliceBlockIds) { + const names = new Set(); + const ids = new Set(); + if (!Array.isArray(sliceBlockIds) || sliceBlockIds.length === 0) return { names, ids }; + const idList = sliceBlockIds.map((id) => `'${id}'`).join(', '); + const nameRows = await cypherRows( + backend, + repo, + `MATCH (b:BasicBlock) WHERE b.id IN [${idList}] RETURN b.callees AS callees`, + ); + for (const r of nameRows) { + for (const n of String(r.callees ?? '').split(' ')) if (n) names.add(n); + } + try { + const idRows = await cypherRows( + backend, + repo, + `MATCH (b:BasicBlock) WHERE b.id IN [${idList}] RETURN b.calleeIds AS calleeIds`, + ); + for (const r of idRows) { + for (const i of String(r.calleeIds ?? '').split(' ')) + if (i && i !== CALLEES_TRUNCATED_SENTINEL) ids.add(i); + } + } catch { + // pre-v3 index: no `calleeIds` column — leave ids empty (graceful degrade). + } + return { names, ids }; +} + +async function run() { + const argv = process.argv.slice(2); + const repo = readOption(argv, 'repo', 'GitNexus'); + const sample = Math.max(1, Number(readOption(argv, 'sample', '120'))); + const minBlocks = Math.max(2, Number(readOption(argv, 'min-blocks', '6'))); + const src = readOption(argv, 'src', 'gitnexus/src/'); + const depth = Math.max(1, Number(readOption(argv, 'depth', '3'))); + const limit = Math.max(1, Number(readOption(argv, 'limit', '200'))); + const json = hasFlag(argv, 'json'); + + const { LocalBackend } = await import( + path.join(REPO_ROOT, 'src', 'mcp', 'local', 'local-backend.ts') + ); + const backend = new LocalBackend(); + const initialized = await backend.init(); + if (!initialized) + throw new Error('no indexed repositories found; run gitnexus analyze --pdg first'); + + try { + const candidateQuery = (label) => + `MATCH (f:${label}) WHERE f.filePath STARTS WITH '${src}' AND f.endLine > f.startLine + 18 ` + + `RETURN f.name AS name, f.filePath AS filePath, f.startLine AS startLine, ` + + `f.endLine AS endLine, '${label}' AS kind`; + let candidates = [ + ...(await cypherRows(backend, repo, candidateQuery('Function'))), + ...(await cypherRows(backend, repo, candidateQuery('Method'))), + ].filter((c) => c.name && /^[A-Za-z_$][\w$]*$/.test(c.name)); + const stride = Math.max(1, Math.floor(candidates.length / (sample * 3))); + candidates = candidates.filter((_, i) => i % stride === 0); + + const cases = []; + let degraded = 0; + for (const c of candidates) { + if (cases.length >= sample) break; + const lo = Number(c.startLine); + const hi = Number(c.endLine); + if (!Number.isFinite(lo) || !Number.isFinite(hi)) continue; + + const blockRows = await cypherRows( + backend, + repo, + `MATCH (b:BasicBlock) WHERE b.filePath = '${c.filePath}' AND b.startLine >= ${lo} ` + + `AND b.startLine <= ${hi + 1} RETURN b.id AS id, b.startLine AS startLine, ` + + `b.callees AS callees ORDER BY b.startLine`, + ); + const fnLine1b = String(lo + 1); + const own = blockRows.filter((r) => { + const parts = r.id.split(':'); + return parts[parts.length - 3] === fnLine1b; + }); + // Union of all leaf call names across the function's own blocks — the + // complete set name-matching could ever prove. A reached direct callee whose + // resolved (definition) name is NOT in here can NEVER be name-proven: it is + // called via an alias/rename, dynamically, or as a filtered member-read — + // the false-negative (import-alias) surface. + const blockLeafUnion = new Set(); + for (const r of own) { + for (const n of String(r.callees ?? '').split(' ')) + if (n && n !== '*') blockLeafUnion.add(n); + } + const bodyBlocks = own.length; + if (bodyBlocks < minBlocks) continue; + const startLines = own + .map((r) => Number(r.startLine)) + .filter(Number.isFinite) + .sort((a, b) => a - b); + const anchor = startLines[Math.max(1, Math.floor(bodyBlocks / 3))]; + if (!Number.isFinite(anchor)) continue; + + const pdg = await backend.callTool('impact', { + repo, + target: c.name, + file_path: c.filePath, + kind: c.kind, + direction: 'downstream', + maxDepth: depth, + limit, + includeTests: true, + mode: 'pdg', + line: anchor, + }); + if (pdg?.error) continue; + if (pdg?.pdgLayer && pdg.pdgLayer !== 'ready') { + degraded++; + continue; + } + if (pdg?.epistemic === 'pdg-no-block-at-line') continue; + + const spByDepth = pdg?.pdgInterprocedural?.statementPreciseByDepth ?? {}; + const d1 = nameCollisionStats(provenItemsAtDepth(spByDepth, 1)); + // All-depth (depth-1 firing + inherited deeper) for context only. + const allProven = Object.keys(spByDepth).flatMap((d) => + provenItemsAtDepth(spByDepth, Number(d)), + ); + const all = nameCollisionStats(allProven); + + // FN / aliasing axis: depth-1 reached direct callees (proven + unproven) + // whose resolved name is absent from EVERY block leaf of the function — so + // name-matching can never prove them even if they are on the slice. + const reachedD1 = provGetReachedD1(pdg); + let reachedD1Names = 0; + let divergentReached = 0; + const seenReached = new Set(); + for (const it of reachedD1) { + const nm = it && typeof it.name === 'string' ? it.name : ''; + const id = it && typeof it.id === 'string' ? it.id : `${nm}@?`; + if (!nm || seenReached.has(id)) continue; + seenReached.add(id); + reachedD1Names += 1; + if (!blockLeafUnion.has(nm)) divergentReached += 1; + } + + // ── U8 realized id-vs-name diff on the EXACT slice ────────────────────── + // Both proven sets are computed by the SAME bridge predicate replica + // (`bridgeProvenSets`) over the depth-1 reached callees and the EXACT + // seed∪reachable slice's `callees`/`calleeIds` — so the whole-symbol and + // sentinel fallbacks (which prove ALL reached items identically on both + // sides) cancel, and only the discriminating divergence survives: + // fpEliminated = name-proven ∖ id-proven (collision FP removed), + // fnRecovered = id-proven ∖ name-proven (import-alias FN recovered). + // On a pre-v3 index `calleeIds` is absent → the id set == the name set → + // both diffs are structurally 0 (the honest degraded reading). + const exactSlice = [ + ...(Array.isArray(pdg?.seedBlocks) ? pdg.seedBlocks : []), + ...(Array.isArray(pdg?.reachableBlocks) ? pdg.reachableBlocks : []), + ]; + const { names: sliceCalleeNames, ids: sliceCalleeIds } = await sliceCalleeSetsOf( + backend, + repo, + exactSlice, + ); + const proven = bridgeProvenSets(reachedD1, sliceCalleeNames, sliceCalleeIds); + const idVsName = scoreIdVsName(proven.nameProven, proven.idProven); + + cases.push({ + name: c.name, + kind: c.kind, + file: c.filePath, + anchor, + sliceBlocks: pdg?.affectedStatementCount ?? 0, + statementPrecision: + typeof pdg?.pdgInterprocedural?.statementPrecision === 'number' + ? round(pdg.pdgInterprocedural.statementPrecision) + : null, + // headline = depth 1 (where name-matching actually fires) + provenLabels: d1.provenLabels, + ambiguousLabels: d1.ambiguousLabels, + excessLabels: d1.excessLabels, + topAmbiguous: d1.ambiguousNames.slice(0, 4), + allDepthProven: all.provenLabels, + allDepthAmbiguous: all.ambiguousLabels, + reachedD1: reachedD1Names, + divergentReached, + // U8 exact-slice id-vs-name diff + nameProven: idVsName.nameProven, + idProven: idVsName.idProven, + fpEliminated: idVsName.fpEliminated, + fnRecovered: idVsName.fnRecovered, + fpEliminatedKeys: idVsName.fpEliminatedKeys, + fnRecoveredKeys: idVsName.fnRecoveredKeys, + // True only when names present, no sentinel, and ids present — the regime + // where the id and name bridges can diverge (else both whole-symbol/name + // fall back identically). Lets the summary confirm the diff is concentrated + // on discriminating slices, not a fallback artifact. + discriminatingSlice: proven.discriminating, + }); + } + + const summary = summarize(cases); + const idVsNameSummary = summarizeIdVsName(cases); + const report = { + repo, + direction: 'downstream', + sample: cases.length, + minBlocks, + degradedSkipped: degraded, + generatedAt: new Date().toISOString(), + note: + 'Realized name-collision probe (depth 1). ambiguousLabels = proven labels whose ' + + 'leaf name is shared by >=2 distinct reached symbol-ids — an UPPER BOUND on ' + + 'collision false-positives (complete: every collision-FP is such a label). ' + + 'excessLabels = sum(count-1) per ambiguous name = central FP estimate. U8 ' + + 'idVsName diffs the EXACT-slice name-proven vs id-proven sets: fpEliminated = ' + + 'realized collision FP the id bridge removes; fnRecovered = realized alias FN it ' + + 'recovers. On a v3+ (calleeIds) index ambiguityRate should collapse to ~0.', + summary, + idVsName: idVsNameSummary, + cases, + }; + + if (json) { + process.stdout.write(JSON.stringify(report, null, 2) + '\n'); + return; + } + + const s = summary; + const lines = []; + lines.push('=== impact-PDG realized name-collision probe (depth 1) ==='); + lines.push(`repo ${repo} | downstream | functions ${cases.length} | minBlocks ${minBlocks}`); + lines.push(''); + lines.push( + `Proven labels: ${s.totalProvenLabels} across ${s.functionsWithProvenLabels} functions ` + + `(median ${s.medianProvenPerFn}/fn).`, + ); + lines.push( + `Name-ambiguous proven labels (UPPER BOUND on collision-FP): ${s.totalAmbiguousLabels} ` + + `(${fmt((s.ambiguityRate ?? 0) * 100, 1)}% of proven), in ${s.functionsWithAmbiguity}/` + + `${cases.length} functions.`, + ); + lines.push( + `Excess proven labels (central FP estimate, sum(count-1)): ${s.totalExcessLabels} ` + + `(${fmt((s.excessRate ?? 0) * 100, 1)}% of proven).`, + ); + lines.push( + `FN / aliasing surface: ${s.totalDivergentReached}/${s.totalReachedD1} depth-1 reached ` + + `callees (${fmt((s.divergenceRate ?? 0) * 100, 1)}%) have a resolved name absent from ` + + `every block leaf (alias/rename/dynamic), in ${s.functionsWithDivergence}/${cases.length} ` + + `functions — name-matching can never prove these.`, + ); + lines.push(''); + const v = idVsNameSummary; + lines.push('--- U8 realized id-vs-name diff (exact seed∪reachable slice) ---'); + lines.push( + `Name-proven labels: ${v.totalNameProven} | id-proven labels: ${v.totalIdProven} ` + + `(across ${v.n} functions; ${v.functionsWithDiscriminatingSlice} have a discriminating ` + + `slice where the two bridges can diverge).`, + ); + lines.push( + `fpEliminated (collision FP the id bridge REMOVES, name∖id): ${v.totalFpEliminated} ` + + `(${fmt((v.fpEliminatedRate ?? 0) * 100, 1)}% of name-proven), in ` + + `${v.functionsWithFpEliminated}/${cases.length} functions.`, + ); + lines.push( + `fnRecovered (alias FN the id bridge RECOVERS, id∖name): ${v.totalFnRecovered} ` + + `(${fmt((v.fnRecoveredRate ?? 0) * 100, 1)}% of id-proven), in ` + + `${v.functionsWithFnRecovered}/${cases.length} functions.`, + ); + lines.push(''); + lines.push( + 'Interpretation: ambiguityRate is the fraction of statement-precise proven labels the ' + + 'NAME match cannot disambiguate (>=2 reached callees share the leaf name). On a v3+ ' + + 'index it collapses to ~0 because the id bridge already discriminates same-named ' + + 'callees — that ~0 is itself proof. fpEliminated is the REALIZED collision FP the id ' + + 'bridge proved away on these exact slices; fnRecovered is the realized import-alias FN ' + + 'it recovered. On a pre-v3 index (no calleeIds) both are 0 (id set == name set).', + ); + process.stdout.write(lines.join('\n') + '\n'); + } finally { + await backend.dispose().catch(() => {}); + } +} + +if (path.resolve(process.argv[1] ?? '') === fileURLToPath(import.meta.url)) { + run().catch((err) => { + process.stderr.write(`[impact-pdg-name-collision] ERROR: ${err?.stack || err}\n`); + process.exit(1); + }); +} diff --git a/gitnexus/bench/impact-pdg/real-code.mjs b/gitnexus/bench/impact-pdg/real-code.mjs new file mode 100644 index 000000000..1a888ba7c --- /dev/null +++ b/gitnexus/bench/impact-pdg/real-code.mjs @@ -0,0 +1,457 @@ +/** + * Real-code performance and quality proxy probe for impact modes. + * + * This complements `measure.mjs`, which is the ground-truth accuracy gate over + * curated fixtures. A real repository does not have an AIS annotation set, so + * this probe does NOT claim accuracy. It checks whether unified `mode:'pdg'` + * preserves the established callgraph symbol reach on a real index, how much it + * costs, and how honest its PDG evidence/degraded signals are. + */ +import fs from 'node:fs'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { fileURLToPath } from 'node:url'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const REPO_ROOT = path.resolve(__dirname, '..', '..'); + +export const DEFAULT_REAL_CODE_CASES = [ + { + name: 'cli-format-impact-upstream', + target: 'formatImpactResult', + file_path: 'gitnexus/src/cli/eval-server.ts', + kind: 'Function', + direction: 'upstream', + line: 208, + }, + { + name: 'cli-impact-command-downstream', + target: 'impactCommand', + file_path: 'gitnexus/src/cli/tool.ts', + kind: 'Function', + direction: 'downstream', + // Block-start line of the coalesced backend-call statement group. The CFG + // coalesces lines 162-192 into one BasicBlock, so an anchor mid-block (e.g. + // 173) lands on no block start and degrades to pdg-no-block-at-line. Seed the + // block's start line so the intra slice is exercised on a real statement. + line: 162, + }, + { + name: 'pdg-engine-downstream', + target: 'runImpactPDG', + file_path: 'gitnexus/src/mcp/local/pdg-impact.ts', + kind: 'Function', + direction: 'downstream', + // Block-start line of the function's opening coalesced statement group + // (the destructure + budget setup spanning 912+). Mid-block lines like 952 + // resolve to no block start; 912 seeds a real, statement-rich intra slice. + line: 912, + }, + { + name: 'pdg-dispatch-upstream', + target: '_impactImpl', + file_path: 'gitnexus/src/mcp/local/local-backend.ts', + kind: 'Method', + direction: 'upstream', + line: 4427, + }, + { + name: 'pdg-compose-downstream', + target: 'composeUnifiedPdgImpactResult', + file_path: 'gitnexus/src/mcp/local/local-backend.ts', + kind: 'Method', + direction: 'downstream', + line: 4850, + }, +]; + +function readOption(argv, name, fallback = undefined) { + const eq = argv.find((arg) => arg.startsWith(`--${name}=`)); + if (eq) return eq.slice(name.length + 3); + const idx = argv.indexOf(`--${name}`); + if (idx >= 0 && idx + 1 < argv.length) return argv[idx + 1]; + return fallback; +} + +function hasFlag(argv, name) { + return argv.includes(`--${name}`); +} + +export function median(xs) { + if (xs.length === 0) return null; + const sorted = [...xs].sort((a, b) => a - b); + const mid = Math.floor(sorted.length / 2); + return sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2; +} + +export function percentile(xs, pct) { + if (xs.length === 0) return null; + const sorted = [...xs].sort((a, b) => a - b); + const idx = Math.min(sorted.length - 1, Math.max(0, Math.ceil((pct / 100) * sorted.length) - 1)); + return sorted[idx]; +} + +function round(value, digits = 3) { + if (value === null || value === undefined || Number.isNaN(value)) return null; + const scale = 10 ** digits; + return Math.round(value * scale) / scale; +} + +function fmt(value, digits = 1) { + return value === null || value === undefined ? 'n/a' : Number(value).toFixed(digits); +} + +export function symbolKeysFromByDepth(byDepth) { + const keys = new Set(); + for (const items of Object.values(byDepth ?? {})) { + for (const item of items ?? []) { + if (!item || typeof item !== 'object') continue; + if (typeof item.id === 'string' && item.id.length > 0) { + keys.add(item.id); + continue; + } + const name = typeof item.name === 'string' ? item.name : '(unknown)'; + const filePath = typeof item.filePath === 'string' ? item.filePath : '(unknown)'; + keys.add(`${name}@${filePath}`); + } + } + return keys; +} + +export function compareSymbolSets(reference, candidate) { + const ref = new Set(reference); + const cand = new Set(candidate); + const overlap = [...ref].filter((key) => cand.has(key)); + const referenceOnly = [...ref].filter((key) => !cand.has(key)).sort(); + const candidateOnly = [...cand].filter((key) => !ref.has(key)).sort(); + const unionSize = new Set([...ref, ...cand]).size; + return { + referenceSize: ref.size, + candidateSize: cand.size, + overlapSize: overlap.length, + recallVsReference: ref.size === 0 ? null : overlap.length / ref.size, + precisionVsReference: cand.size === 0 ? null : overlap.length / cand.size, + jaccard: unionSize === 0 ? null : overlap.length / unionSize, + referenceOnly, + candidateOnly, + }; +} + +function sumEvidenceCounts(results) { + const counts = {}; + for (const result of results) { + const evidenceCounts = + result?.pdgInterprocedural?.evidenceCounts ?? + result?.pdgEvidence?.interproceduralEvidenceCounts ?? + {}; + for (const [key, value] of Object.entries(evidenceCounts)) { + counts[key] = (counts[key] ?? 0) + Number(value ?? 0); + } + } + return counts; +} + +function readCases(caseFile) { + if (!caseFile) return DEFAULT_REAL_CODE_CASES; + const resolved = path.resolve(process.cwd(), caseFile); + const parsed = JSON.parse(fs.readFileSync(resolved, 'utf8')); + const cases = Array.isArray(parsed) ? parsed : parsed.cases; + if (!Array.isArray(cases) || cases.length === 0) { + throw new Error(`case file ${resolved} must contain a non-empty array or { "cases": [...] }`); + } + return cases; +} + +async function timedImpact(backend, params) { + const started = performance.now(); + const result = await backend.callTool('impact', params); + return { result, ms: performance.now() - started }; +} + +async function measureCase(backend, testCase, options) { + const baseParams = { + repo: options.repo, + target: testCase.target, + file_path: testCase.file_path, + kind: testCase.kind, + direction: testCase.direction ?? 'upstream', + maxDepth: options.depth, + includeTests: options.includeTests, + limit: options.limit, + }; + const callgraphTimes = []; + const pdgTimes = []; + let callgraphResult = null; + let pdgResult = null; + + for (let i = 0; i < options.repeat; i++) { + const callgraph = await timedImpact(backend, { ...baseParams, mode: 'callgraph' }); + callgraphTimes.push(callgraph.ms); + callgraphResult = callgraph.result; + + const pdg = await timedImpact(backend, { + ...baseParams, + mode: 'pdg', + ...(Number.isInteger(testCase.line) ? { line: testCase.line } : {}), + }); + pdgTimes.push(pdg.ms); + pdgResult = pdg.result; + } + + const callgraphKeys = symbolKeysFromByDepth(callgraphResult?.byDepth ?? {}); + const pdgInterByDepth = + pdgResult?.interproceduralByDepth ?? pdgResult?.pdgInterprocedural?.byDepth ?? {}; + const pdgInterKeys = symbolKeysFromByDepth(pdgInterByDepth); + const symbolAgreement = compareSymbolSets(callgraphKeys, pdgInterKeys); + const evidenceCounts = + pdgResult?.pdgInterprocedural?.evidenceCounts ?? + pdgResult?.pdgEvidence?.interproceduralEvidenceCounts ?? + {}; + + return { + name: testCase.name ?? testCase.target, + target: testCase.target, + filePath: testCase.file_path, + kind: testCase.kind, + direction: baseParams.direction, + line: Number.isInteger(testCase.line) ? testCase.line : null, + latencyMs: { + callgraph: { + median: round(median(callgraphTimes)), + p95: round(percentile(callgraphTimes, 95)), + samples: callgraphTimes.map((v) => round(v)), + }, + pdg: { + median: round(median(pdgTimes)), + p95: round(percentile(pdgTimes, 95)), + samples: pdgTimes.map((v) => round(v)), + }, + pdgOverCallgraphMedian: + median(callgraphTimes) && median(callgraphTimes) > 0 + ? round(median(pdgTimes) / median(callgraphTimes)) + : null, + }, + callgraph: { + error: callgraphResult?.error ?? null, + impactedCount: callgraphResult?.impactedCount ?? 0, + risk: callgraphResult?.risk ?? null, + epistemic: callgraphResult?.epistemic ?? null, + partial: Boolean(callgraphResult?.partial), + symbolCount: callgraphKeys.size, + }, + pdg: { + error: pdgResult?.error ?? null, + pdgLayer: pdgResult?.pdgLayer ?? 'ready', + epistemic: pdgResult?.epistemic ?? null, + partial: Boolean(pdgResult?.partial || pdgResult?.pdgInterprocedural?.partial), + impactedCount: pdgResult?.impactedCount ?? 0, + affectedStatementCount: pdgResult?.affectedStatementCount ?? 0, + blockCount: pdgResult?.blockCount ?? 0, + interproceduralSymbolCount: pdgInterKeys.size, + evidence: pdgResult?.pdgInterprocedural?.evidence ?? pdgResult?.pdgEvidence?.interprocedural, + evidenceCounts, + }, + symbolAgreement, + }; +} + +export function summarizeCases(cases) { + const ratios = cases + .map((c) => c.latencyMs.pdgOverCallgraphMedian) + .filter((v) => v !== null && v !== undefined); + const callgraphMedians = cases + .map((c) => c.latencyMs.callgraph.median) + .filter((v) => v !== null && v !== undefined); + const pdgMedians = cases + .map((c) => c.latencyMs.pdg.median) + .filter((v) => v !== null && v !== undefined); + const comparable = cases.filter((c) => c.symbolAgreement.recallVsReference !== null); + const recalls = comparable.map((c) => c.symbolAgreement.recallVsReference); + const precisions = comparable + .map((c) => c.symbolAgreement.precisionVsReference) + .filter((v) => v !== null && v !== undefined); + const degradedCases = cases.filter((c) => c.pdg.pdgLayer !== 'ready'); + const errorCases = cases.filter((c) => c.callgraph.error || c.pdg.error); + const partialCases = cases.filter((c) => c.callgraph.partial || c.pdg.partial); + const noBlockAtLineCases = cases.filter((c) => c.pdg.epistemic === 'pdg-no-block-at-line'); + const evidenceCounts = sumEvidenceCounts(cases.map((c) => ({ pdgInterprocedural: c.pdg }))); + const totalBridgeSymbols = Object.values(evidenceCounts).reduce((a, b) => a + Number(b ?? 0), 0); + + return { + performance: { + callgraphMedianMs: round(median(callgraphMedians)), + pdgMedianMs: round(median(pdgMedians)), + pdgP95Ms: round(percentile(cases.map((c) => c.latencyMs.pdg.p95).filter(Boolean), 95)), + pdgOverCallgraphMedian: round(median(ratios)), + }, + qualityProxy: { + comparableCases: comparable.length, + meanSymbolRecallVsCallgraph: recalls.length + ? round(recalls.reduce((a, b) => a + b, 0) / recalls.length) + : null, + minSymbolRecallVsCallgraph: recalls.length ? round(Math.min(...recalls)) : null, + meanSymbolPrecisionVsCallgraph: precisions.length + ? round(precisions.reduce((a, b) => a + b, 0) / precisions.length) + : null, + degradedCaseCount: degradedCases.length, + errorCaseCount: errorCases.length, + partialCaseCount: partialCases.length, + noBlockAtLineCaseCount: noBlockAtLineCases.length, + evidenceCounts, + unprovenBridgeRatio: + totalBridgeSymbols > 0 + ? round((evidenceCounts['unproven-bridge'] ?? 0) / totalBridgeSymbols) + : null, + }, + }; +} + +export function evaluateCheckGates(report, env = process.env) { + const failures = []; + const minRecall = Number(env.GN_REAL_CODE_PDG_MIN_SYMBOL_RECALL ?? 0.95); + const maxMedianMs = Number(env.GN_REAL_CODE_PDG_MAX_MEDIAN_MS ?? 5000); + const quality = report.summary.qualityProxy; + const perf = report.summary.performance; + + if (report.cases.length === 0) failures.push('no real-code cases were measured'); + if (quality.errorCaseCount > 0) + failures.push(`${quality.errorCaseCount} case(s) returned errors`); + if (quality.degradedCaseCount > 0) { + failures.push(`${quality.degradedCaseCount} case(s) reported a degraded PDG layer`); + } + if ( + quality.minSymbolRecallVsCallgraph !== null && + quality.minSymbolRecallVsCallgraph < minRecall + ) { + failures.push( + `min PDG symbol recall vs callgraph ${quality.minSymbolRecallVsCallgraph} < ${minRecall}`, + ); + } + if (perf.pdgMedianMs !== null && perf.pdgMedianMs > maxMedianMs) { + failures.push(`PDG median latency ${perf.pdgMedianMs}ms > ${maxMedianMs}ms`); + } + return failures; +} + +function renderText(report, failures) { + const lines = []; + const perf = report.summary.performance; + const quality = report.summary.qualityProxy; + lines.push('=== impact-PDG real-code performance/quality probe ==='); + lines.push( + `repo ${report.repo} | cases ${report.cases.length} | repeat ${report.repeat} | includeTests=${report.includeTests}`, + ); + lines.push(''); + lines.push( + `Latency: callgraph median ${fmt(perf.callgraphMedianMs)}ms, ` + + `pdg median ${fmt(perf.pdgMedianMs)}ms, pdg p95 ${fmt(perf.pdgP95Ms)}ms, ` + + `median overhead ${fmt(perf.pdgOverCallgraphMedian, 2)}x`, + ); + lines.push( + `Quality proxy: min PDG symbol recall vs callgraph ${fmt( + quality.minSymbolRecallVsCallgraph, + 3, + )}, mean recall ${fmt(quality.meanSymbolRecallVsCallgraph, 3)}, ` + + `mean precision ${fmt(quality.meanSymbolPrecisionVsCallgraph, 3)}`, + ); + lines.push( + `Signals: degraded=${quality.degradedCaseCount}, errors=${quality.errorCaseCount}, ` + + `partial=${quality.partialCaseCount}, no-block-at-line=${quality.noBlockAtLineCaseCount}, ` + + `unprovenBridgeRatio=${fmt(quality.unprovenBridgeRatio, 3)}`, + ); + lines.push(`Evidence counts: ${JSON.stringify(quality.evidenceCounts)}`); + lines.push(''); + lines.push('Per case:'); + for (const c of report.cases) { + lines.push( + ` ${c.name}: cg ${fmt(c.latencyMs.callgraph.median)}ms/${c.callgraph.symbolCount} symbols, ` + + `pdg ${fmt(c.latencyMs.pdg.median)}ms/${c.pdg.interproceduralSymbolCount} inter-symbols, ` + + `statements=${c.pdg.affectedStatementCount}, recall=${fmt( + c.symbolAgreement.recallVsReference, + 3, + )}, precision=${fmt(c.symbolAgreement.precisionVsReference, 3)}, ` + + `evidence=${c.pdg.evidence ?? 'n/a'}`, + ); + if (c.callgraph.error || c.pdg.error || c.pdg.pdgLayer !== 'ready') { + lines.push( + ` status: callgraphError=${c.callgraph.error ?? 'none'} pdgError=${ + c.pdg.error ?? 'none' + } pdgLayer=${c.pdg.pdgLayer}`, + ); + } + if (c.symbolAgreement.referenceOnly.length > 0) { + lines.push( + ` callgraph-only symbols: ${c.symbolAgreement.referenceOnly.slice(0, 5).join(', ')}`, + ); + } + } + lines.push(''); + lines.push( + 'Interpretation: this real-code probe measures latency and quality proxies, not accuracy. ' + + 'The curated fixture harness remains the AIS-backed accuracy gate.', + ); + if (failures.length > 0) { + lines.push(''); + for (const failure of failures) lines.push(`[impact-pdg-real-code --check] FAIL: ${failure}`); + } + return lines.join('\n'); +} + +async function run() { + const argv = process.argv.slice(2); + const repo = readOption(argv, 'repo', 'GitNexus'); + const repeat = Math.max(1, Number(readOption(argv, 'repeat', '3'))); + const depth = Math.max(1, Number(readOption(argv, 'depth', '3'))); + const limit = Math.max(1, Number(readOption(argv, 'limit', '100'))); + const includeTests = readOption(argv, 'include-tests', 'true') !== 'false'; + const caseFile = readOption(argv, 'case-file'); + const json = hasFlag(argv, 'json'); + const check = hasFlag(argv, 'check'); + const cases = readCases(caseFile); + + const { LocalBackend } = await import( + path.join(REPO_ROOT, 'src', 'mcp', 'local', 'local-backend.ts') + ); + const backend = new LocalBackend(); + const initialized = await backend.init(); + if (!initialized) + throw new Error('no indexed repositories found; run gitnexus analyze --pdg first'); + + try { + const measured = []; + for (const testCase of cases) { + measured.push( + await measureCase(backend, testCase, { repo, repeat, depth, limit, includeTests }), + ); + } + const report = { + repo, + repeat, + depth, + limit, + includeTests, + generatedAt: new Date().toISOString(), + note: 'Real-code probe: latency plus quality proxies only. Accuracy requires AIS-backed fixtures.', + cases: measured, + summary: summarizeCases(measured), + }; + const failures = evaluateCheckGates(report); + + if (json) { + process.stdout.write(JSON.stringify({ ...report, checkFailures: failures }, null, 2) + '\n'); + } else { + process.stdout.write(renderText(report, failures) + '\n'); + } + + if (check && failures.length > 0) process.exit(1); + } finally { + await backend.dispose().catch(() => {}); + } +} + +if (path.resolve(process.argv[1] ?? '') === fileURLToPath(import.meta.url)) { + run().catch((err) => { + process.stderr.write(`[impact-pdg-real-code] ERROR: ${err?.stack || err}\n`); + process.exit(1); + }); +} diff --git a/gitnexus/src/cli/ai-context.ts b/gitnexus/src/cli/ai-context.ts index 8cbfe2b6d..299b4b7a3 100644 --- a/gitnexus/src/cli/ai-context.ts +++ b/gitnexus/src/cli/ai-context.ts @@ -199,7 +199,11 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s ## Always Do -- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run \`impact({target: "symbolName", direction: "upstream"})\` and report the blast radius (direct callers, affected processes, risk level) to the user. +- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run \`impact({target: "symbolName", direction: "upstream"})\` and report the blast radius (direct callers, affected processes, risk level) to the user.${ + hasPdg + ? ` For unified PDG impact, add \`mode: "pdg"\` with optional \`line: \` — it returns statement-level \`affectedStatements\` over CDG + REACHING_DEF and inter-procedural symbols in \`interproceduralByDepth\`/\`byDepth\`; no-layer/degraded PDG results are UNKNOWN-risk notes (\`--pdg\` layer).` + : '' + } - **MUST run \`detect_changes()\` before committing** to verify your changes only affect expected symbols and execution flows. For regression review, compare against the default branch: \`detect_changes({scope: "compare", base_ref: ${JSON.stringify(markdownSafeBranch(defaultBranch))}})\`. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. - When exploring unfamiliar code, use \`query({search_query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. diff --git a/gitnexus/src/cli/eval-server.ts b/gitnexus/src/cli/eval-server.ts index a3a058264..ef900a8c6 100644 --- a/gitnexus/src/cli/eval-server.ts +++ b/gitnexus/src/cli/eval-server.ts @@ -30,6 +30,7 @@ */ import http from 'http'; +import crypto from 'node:crypto'; import { isIPv4, isIPv6 } from 'node:net'; import { writeSync } from 'node:fs'; import { @@ -178,6 +179,18 @@ export function formatContextResult(result: any): string { return lines.join('\n').trim(); } +function formatTruncationSuffix(result: { + truncatedBy?: unknown; + truncatedByReasons?: unknown; +}): string { + const label = Array.isArray(result.truncatedByReasons) + ? result.truncatedByReasons.join(', ') + : typeof result.truncatedBy === 'string' + ? result.truncatedBy + : ''; + return label ? ` (by ${label})` : ''; +} + export function formatImpactResult(result: any): string { if (result.error) { const suggestion = result.suggestion ? `\nSuggestion: ${result.suggestion}` : ''; @@ -194,6 +207,28 @@ export function formatImpactResult(result: any): string { // mirroring formatContextResult, so the real impact under whichever symbol the // caller meant is visible on the text surface, not just in the JSON. if (result.status === 'ambiguous') { + if (result.mode === 'pdg') { + const shown = result.candidates?.length ?? 0; + const totalCandidates = result.totalCandidates ?? shown; + const countPhrase = + totalCandidates > shown + ? `${totalCandidates} symbols (showing ${shown})` + : `${totalCandidates} symbols`; + const lines = [ + `${target?.name || '?'}: AMBIGUOUS — ${countPhrase} share this name. ` + + `PDG impact was not computed until the target is disambiguated. ` + + `Use --uid, file_path, or kind for one authoritative PDG result.`, + ]; + if (result.message) lines.push(String(result.message)); + for (const c of result.candidates || []) { + const score = typeof c.score === 'number' ? ` score ${c.score}` : ''; + lines.push( + ` ${c.kind} ${c.name} → ${c.filePath}:${c.line || '?'}${score} (uid: ${c.uid})`, + ); + } + return lines.join('\n'); + } + // #2129 review F11 — report the FULL match count (`totalCandidates`), not the // truncated `candidates[]` length; note when the candidate list is capped. const shown = result.candidates?.length ?? 0; @@ -219,6 +254,204 @@ export function formatImpactResult(result: any): string { return lines.join('\n'); } + // ─── PDG mode (mode:'pdg') ──────────────────────────────────────────── + // KTD8 presentation half. PDG results are intra-procedural Program + // Dependence Graph blast radii: the single collapsed `byDepth[1]` bucket + // has NO call-hop depth meaning (block-hops ≠ call-hops), so we must NOT + // reuse the callgraph "depth N / WILL BREAK (direct)" framing, the + // callgraph DI/dynamic-dispatch lower-bound copy, or the confident + // "isolated" zero. A degraded / no-body PDG result is INCONCLUSIVE, not + // safe-to-refactor — it gets the explicit caveat + remediation, never an + // empty blast radius. Detect on `mode:'pdg'` (every PDG return path — + // findings, degradation, no-body, no-dependence — carries it). Ambiguous + // PDG results carry `status:'ambiguous'` and are handled above; they never + // reach here. + if (result.mode === 'pdg') { + const name = target?.name || '?'; + const appendPdgInterproceduralSymbols = (lines: string[]): boolean => { + const byDepth = + result.interproceduralByDepth || result.pdgInterprocedural?.byDepth || result.byDepth || {}; + const byDepthCounts = + result.interproceduralByDepthCounts || + result.pdgInterprocedural?.byDepthCounts || + result.byDepthCounts || + {}; + const depthKeys = Array.from( + new Set([...Object.keys(byDepthCounts), ...Object.keys(byDepth)]), + ) + .map((d) => Number(d)) + .filter((d) => Number.isFinite(d)) + .sort((a, b) => a - b); + const hasReach = depthKeys.some((depth) => { + const items = byDepth[depth] || byDepth[String(depth)] || []; + const count = byDepthCounts[depth] ?? byDepthCounts[String(depth)] ?? items.length; + return count > 0; + }); + if (!hasReach) return false; + + const totalSymbols = + result.pdgInterprocedural?.impactedCount ?? + (typeof result.impactedCount === 'number' ? result.impactedCount : 0); + lines.push(''); + lines.push(`Inter-procedural symbol reach (${totalSymbols}):`); + for (const depth of depthKeys) { + const items = byDepth[depth] || byDepth[String(depth)] || []; + const count = byDepthCounts[depth] ?? byDepthCounts[String(depth)] ?? items.length; + if (count <= 0) continue; + lines.push(` d=${depth} (${count})`); + const shown = Math.min(items.length, 12); + for (const item of items.slice(0, shown)) { + const flags: string[] = []; + if (item.unresolved) flags.push('unresolved'); + if (item.ambiguous) flags.push('ambiguous'); + const flagStr = flags.length ? ` [${flags.join(', ')}]` : ''; + lines.push(` ${item.type || ''} ${item.name} → ${item.filePath}${flagStr}`); + } + if (count > shown) lines.push(` ... and ${count - shown} more`); + } + return true; + }; + + // (1) Degradation — the PDG layer (or a sub-layer) is absent/unreadable. + // `pdgLayer` is the non-'ready' state from `pdgLayerStatus`. Print the + // honest remediation, NOT a zero/empty blast radius. + if (result.pdgLayer) { + const subLayer = result.missingSubLayer + ? ` (missing sub-layer: ${result.missingSubLayer})` + : ''; + return ( + `${name}: PDG impact unavailable — the index has no usable PDG layer ` + + `[${result.pdgLayer}]${subLayer}. This is NOT "no impact". ` + + `Re-index with \`gitnexus analyze --pdg\` to build the control/data ` + + `dependence layer, or use \`--mode callgraph\` for the call-graph blast radius.` + + (result.note ? `\n${result.note}` : '') + ); + } + + // (2) No-body symbol (KTD6) — interface / type alias / abstract / ambient + // member / one-line declaration with no CFG. Show the caveat, never + // "isolated / no dependencies". + if (result.epistemic === 'no-pdg-body') { + const noBodyLines = [ + `${name}: local PDG slice not applicable to this symbol — it has no PDG body ` + + `(no control/data dependence edges; e.g. an interface, type alias, ` + + `abstract/ambient member, or a one-line declaration). This is NOT a ` + + `confident "no impact".`, + ]; + appendPdgInterproceduralSymbols(noBodyLines); + if (result.note) noBodyLines.push(result.note); + return noBodyLines.join('\n'); + } + + // (2b) STATEMENT-ANCHORED SLICE (mode:'pdg' + line). When `criterionLine` is + // present the result is a statement slice: the seeded line plus the list of + // dependent statements (`affectedStatements: {line,filePath,text}[]`). Render + // those statements directly — this IS the useful output of statement mode — + // rather than the symbol-projection bucket below. Empty cases: + // - `pdg-no-block-at-line`: the line is blank / a comment / outside the + // body (no statement block) — print the steering note. + // - empty `affectedStatements` with `pdg-intra-procedural`: the line has no + // dependents in this direction — print the steering note. + // Each non-empty case also surfaces truncation honestly. + if (typeof result.criterionLine === 'number') { + const slice: any[] = Array.isArray(result.affectedStatements) + ? result.affectedStatements + : []; + const count = + typeof result.affectedStatementCount === 'number' + ? result.affectedStatementCount + : slice.length; + // File anchor for the heading — the seeded statement's file (every slice + // statement shares the function's file). Fall back to the target's file. + const anchorFile = slice[0]?.filePath || target?.filePath || name; + + if (count === 0 || slice.length === 0) { + // No statement block at the line, or no dependents in this direction. + // Print the honest note (pdg-no-block-at-line or the no-dependence note) + // verbatim — never an empty "isolated" headline. + const emptySliceLines = [ + `No statements ${direction}-dependent on ${anchorFile}:${result.criterionLine}.`, + ]; + if (result.truncated) { + const by = formatTruncationSuffix(result); + emptySliceLines.push( + `⚠️ Truncated${by} — the dependence slice was bounded; deeper PDG-dependent statements may exist.`, + ); + } + appendPdgInterproceduralSymbols(emptySliceLines); + if (result.note) emptySliceLines.push(result.note); + return emptySliceLines.join('\n'); + } + + const slLines: string[] = []; + slLines.push( + `Statements ${direction}-dependent on ${anchorFile}:${result.criterionLine} (${count}):`, + ); + for (const s of slice) { + const text = typeof s.text === 'string' ? s.text : ''; + slLines.push(` L${s.line}: ${text}`); + } + // Truncation honesty — the slice may be a lower bound (depth or per-step + // LIMIT bound). Surface it the same way the symbol render does. + if (result.truncated) { + const by = formatTruncationSuffix(result); + slLines.push( + `⚠️ Truncated${by} — the dependence slice was bounded; deeper PDG-dependent statements may exist.`, + ); + } + appendPdgInterproceduralSymbols(slLines); + if (result.note) { + slLines.push(''); + slLines.push(`ℹ️ ${result.note}`); + } + return slLines.join('\n').trim(); + } + + const pdgLines: string[] = []; + + if (!appendPdgInterproceduralSymbols(pdgLines)) { + pdgLines.push( + `${name} (${direction}): no inter-procedural symbols reached. ` + + `The local PDG statement slice may still report affectedStatements when seeded with line:.`, + ); + } + + // The assembled note carries the local-PDG framing plus the unified + // inter-procedural symbol-reach contract; surface it verbatim so the CLI + // reader sees the same honesty the JSON consumer does. + if (result.note) { + pdgLines.push(''); + pdgLines.push(`ℹ️ ${result.note}`); + } else { + pdgLines.push(''); + pdgLines.push( + 'ℹ️ Program Dependence Graph result — statement reach is reported in affectedStatements and inter-procedural symbol reach in interproceduralByDepth/byDepth.', + ); + } + + // Honest incompleteness signals (block-attribution + truncation). + if (result.ambiguousProjectionCount > 0) { + pdgLines.push( + `⚠️ ${result.ambiguousProjectionCount} block(s) could not be attributed to a ` + + `unique owning symbol (same-line functions) — all colliding symbols are shown.`, + ); + } + if (result.unresolvedBlockCount > 0) { + pdgLines.push( + `⚠️ ${result.unresolvedBlockCount} dependence block(s) map to no owning ` + + `Function/Method/Constructor (top-level statement / closure) — surfaced under their file.`, + ); + } + if (result.truncated) { + const by = formatTruncationSuffix(result); + pdgLines.push( + `⚠️ Truncated${by} — the dependence traversal was bounded; deeper PDG impacts may exist.`, + ); + } + + return pdgLines.join('\n').trim(); + } + if (total === 0) { // #1858 — "isolated" is a confident claim. If an interface / indirection // boundary is on the path, the true count is a lower bound, not zero; @@ -376,7 +609,7 @@ function formatToolResult(toolName: string, result: any): string { // Guide the agent to the logical next tool call. // Critical for tool chaining: query → context → impact → fix. -function getNextStepHint(toolName: string): string { +export function getNextStepHint(toolName: string, result?: any): string { switch (toolName) { case 'query': return '\n---\nNext: Pick a symbol above and run gitnexus-context "" to see all its callers, callees, and execution flows.'; @@ -385,6 +618,15 @@ function getNextStepHint(toolName: string): string { return '\n---\nNext: To check what breaks if you change this, run gitnexus-impact "" upstream'; case 'impact': + if ( + result?.error || + result?.status === 'ambiguous' || + result?.mode === 'pdg' || + result?.pdgLayer || + typeof result?.criterionLine === 'number' + ) { + return ''; + } return '\n---\nNext: Review d=1 items first (WILL BREAK). Read the source with cat to understand the code, then make your fix.'; case 'cypher': @@ -454,6 +696,14 @@ export async function evalServerCommand(options?: EvalServerOptions): Promise { resetIdleTimer(); @@ -468,6 +718,12 @@ export async function evalServerCommand(options?: EvalServerOptions): Promise = {}; @@ -500,7 +764,7 @@ export async function evalServerCommand(options?: EvalServerOptions): Promise` from reaching those through this Docker/eval-harness server. + */ +export const EVAL_SERVER_TOOLS: ReadonlySet = new Set([ + 'query', + 'context', + 'impact', + 'cypher', + 'detect_changes', + 'list_repos', +]); + export const MAX_BODY_SIZE = 1024 * 1024; // 1MB function readBody(req: http.IncomingMessage): Promise { diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index 4ac78ae85..72e4da365 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -338,6 +338,15 @@ program .command('impact [target]') .description('Blast radius analysis: what breaks if you change a symbol') .option('-d, --direction ', 'upstream (dependants) or downstream (dependencies)', 'upstream') + .option( + '--mode ', + 'Engine: callgraph (default) or pdg (opt-in, intra-procedural; needs analyze --pdg)', + 'callgraph', + ) + .option( + '--line ', + '1-based source line — PDG-only statement anchor (--mode pdg): slice the dependence from the statement at this line and show what depends on it', + ) .option('-r, --repo ', 'Target repository') .option('--branch ', 'Scope to a specific branch index (multi-branch repos)') .option('-u, --uid ', 'Direct symbol UID (zero-ambiguity lookup)') diff --git a/gitnexus/src/cli/tool.ts b/gitnexus/src/cli/tool.ts index 111bb4bbc..34b633956 100644 --- a/gitnexus/src/cli/tool.ts +++ b/gitnexus/src/cli/tool.ts @@ -124,6 +124,8 @@ export async function impactCommand( target?: string, options?: { direction?: string; + mode?: string; + line?: string; repo?: string; branch?: string; uid?: string; @@ -162,12 +164,23 @@ export async function impactCommand( const rawOffset = parseInt(options?.offset ?? '', 10); const parsedLimit = Number.isFinite(rawLimit) ? rawLimit : undefined; const parsedOffset = Number.isFinite(rawOffset) ? rawOffset : undefined; + // `--line` is a PDG-only statement anchor (1-based source line). Parse it to + // an integer when provided and thread it ONLY when present, so the backend's + // line-without-pdg / non-positive-integer validation fires on the real value + // rather than on a silently-dropped flag. A non-numeric `--line` parses to + // NaN, which the backend rejects as a non-positive integer (loud, not silent). + const parsedLine = options?.line !== undefined ? parseInt(options.line, 10) : undefined; const result = await backend.callTool('impact', { target: target || undefined, target_uid: options?.uid, file_path: options?.file, kind: options?.kind, direction: options?.direction || 'upstream', + // Forward the engine selector; backend validates the enum (callgraph/pdg) + // and treats the default 'callgraph' identically to an omitted mode. + mode: options?.mode, + // PDG-only statement anchor — forwarded only when --line was given. + ...(parsedLine !== undefined ? { line: parsedLine } : {}), maxDepth: options?.depth ? parseInt(options.depth, 10) : undefined, includeTests: options?.includeTests ?? false, repo: options?.repo, diff --git a/gitnexus/src/core/group/cross-impact.ts b/gitnexus/src/core/group/cross-impact.ts index e8db52dd5..0d7e2f360 100644 --- a/gitnexus/src/core/group/cross-impact.ts +++ b/gitnexus/src/core/group/cross-impact.ts @@ -339,7 +339,11 @@ function extractProcessNames(impact: unknown): string[] { return o.affected_processes.map((p) => String(p.name ?? '')).filter(Boolean); } -function mergeRisk(localRisk: string, cross: CrossRepoImpact[]): string { +// Exported so the U4 PDG-result interchangeability contract (KTD8) can assert +// permanently that a PDG `risk:'UNKNOWN'` never coalesces to a confident `LOW`. +// No behavior change — `'UNKNOWN'` was already handled correctly at the +// `(localRisk === 'LOW' || localRisk === 'UNKNOWN')` branch below. +export function mergeRisk(localRisk: string, cross: CrossRepoImpact[]): string { const highConf = cross.some((c) => c.contract.confidence >= 0.85); if (localRisk === 'CRITICAL') return 'CRITICAL'; if (cross.length >= 3) return 'CRITICAL'; diff --git a/gitnexus/src/core/group/extractors/http-route-extractor.ts b/gitnexus/src/core/group/extractors/http-route-extractor.ts index e14b9b701..450cf3841 100644 --- a/gitnexus/src/core/group/extractors/http-route-extractor.ts +++ b/gitnexus/src/core/group/extractors/http-route-extractor.ts @@ -6,6 +6,7 @@ import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js import type { ExtractedContract, RepoHandle } from '../types.js'; import { readSafe } from './fs-utils.js'; import { parseSourceSafe } from '../../tree-sitter/safe-parse.js'; +import { logger } from '../../logger.js'; import { getPluginForFile, HTTP_SCAN_GLOB, @@ -40,13 +41,16 @@ import { // ─── Graph-assisted queries ────────────────────────────────────────── -const HANDLES_ROUTE_QUERY = ` +// Exported so integration tests can run the exact production query against a +// real LadybugDB (guards the Route.method column contract — see +// route-method-roundtrip.test.ts). +export const HANDLES_ROUTE_QUERY = ` MATCH (handlerFile:File)-[r:CodeRelation {type: 'HANDLES_ROUTE'}]->(route:Route) RETURN handlerFile.id AS fileId, handlerFile.filePath AS filePath, route.name AS routePath, route.id AS routeId, + route.method AS routeMethod, route.responseKeys AS responseKeys, r.reason AS routeSource`; - const FETCHES_QUERY = ` MATCH (callerFile:File)-[r:CodeRelation {type: 'FETCHES'}]->(route:Route) RETURN callerFile.id AS fileId, callerFile.filePath AS filePath, @@ -324,7 +328,16 @@ export class HttpRouteExtractor implements ContractExtractor { let rows: Record[]; try { rows = await db(HANDLES_ROUTE_QUERY); - } catch { + } catch (err) { + // A failure here silently disables the entire graph-assisted HTTP + // provider path (the source-scan fallback still runs and masks most + // of the damage), so surface it at debug level to make a total + // outage observable instead of invisible. + logger.debug( + `[http-route-extractor] HANDLES_ROUTE query failed; graph providers skipped: ${ + err instanceof Error ? err.message : String(err) + }`, + ); return []; } @@ -332,7 +345,14 @@ export class HttpRouteExtractor implements ContractExtractor { const filePath = String(row.filePath ?? ''); const routePath = String(row.routePath ?? ''); const routeSource = String(row.routeSource ?? row.routeReason ?? ''); - let method = methodFromRouteReason(routeSource); + // Prefer the HTTP verb persisted on the Route node by the ingestion + // routes phase (Spring/Laravel framework routes and decorator routes + // carry it). Fall back to parsing it out of the edge reason for + // older indexes or filesystem routes that never stored a method. + const graphMethod = String(row.routeMethod ?? '') + .trim() + .toUpperCase(); + let method = (graphMethod || null) ?? methodFromRouteReason(routeSource); // Look up handler name (and backfill method if missing) from the // plugin's scan of the handler file. This replaces the old @@ -458,7 +478,12 @@ export class HttpRouteExtractor implements ContractExtractor { let rows: Record[]; try { rows = await db(FETCHES_QUERY); - } catch { + } catch (err) { + logger.debug( + `[http-route-extractor] FETCHES query failed; graph consumers skipped: ${ + err instanceof Error ? err.message : String(err) + }`, + ); return []; } for (const row of rows) { diff --git a/gitnexus/src/core/incremental/subgraph-extract.ts b/gitnexus/src/core/incremental/subgraph-extract.ts index 74f01fd48..523d25dd1 100644 --- a/gitnexus/src/core/incremental/subgraph-extract.ts +++ b/gitnexus/src/core/incremental/subgraph-extract.ts @@ -62,7 +62,14 @@ const isGraphWide = (label: string): boolean => label === 'Community' || label = * A→C edge. These are always extracted (and the orchestrator delete-alls them * first, like Community/Process) so they rebuild from the fresh graph. */ -const isGraphWideRelType = (type: string): boolean => type === 'TAINT_PATH'; +// `CALL_SUMMARY` (PDG FU-C) is intra-procedural (a callee's RETURN-VALUE ASCENT +// depends only on its OWN body), but the orchestrator delete-alls it on an +// incremental `--pdg` writeback to keep the emit path single — so it must be +// re-included from the FULL fresh graph (which the emit phase recomputes every +// run) or an unchanged function's summary would be lost. Cheap: one self-loop +// edge per return-flowing function. +const isGraphWideRelType = (type: string): boolean => + type === 'TAINT_PATH' || type === 'CALL_SUMMARY'; /** * Build a Map for every File-bound node in the graph. diff --git a/gitnexus/src/core/ingestion/cfg/emit.ts b/gitnexus/src/core/ingestion/cfg/emit.ts index f4550951d..2bc0f8f77 100644 --- a/gitnexus/src/core/ingestion/cfg/emit.ts +++ b/gitnexus/src/core/ingestion/cfg/emit.ts @@ -20,7 +20,7 @@ */ import type { KnowledgeGraph } from '../../graph/types.js'; import { generateId } from '../../../lib/utils.js'; -import { computeReachingDefs } from './reaching-defs.js'; +import { computeReachingDefs, type ReachingDefsSolver } from './reaching-defs.js'; import { computeControlDependence } from './control-dependence.js'; import { computePostDominators, @@ -28,7 +28,37 @@ import { NO_IPDOM, } from './post-dominators.js'; import { augmentForPostDom } from './synthetic-escape.js'; -import type { BindingEntry, FunctionCfg } from './types.js'; +import { DEFAULT_PDG_MAX_SITES_PER_STATEMENT } from './visitors/call-site-harvest.js'; +import { calleeIdPosKey } from '../scope-resolution/graph-bridge/callee-id-sink.js'; +import { encodeReachingDefReasonPairs } from './reaching-def-reason-codec.js'; +import type { BasicBlockData, BindingEntry, FunctionCfg } from './types.js'; + +/** + * Reserved token placed in `BasicBlock.callees` when a statement's call sites + * were truncated at {@link DEFAULT_PDG_MAX_SITES_PER_STATEMENT}: the recorded + * callee list is then INCOMPLETE, so over-cap callees are absent. `*` is not a + * valid identifier leaf, so it cannot collide with a real callee name. The + * impact bridge treats a slice containing this sentinel as "callees unknown" and + * keeps reach callgraph-equal (proven), rather than falsely labeling an + * absent-but-real callee `unproven-bridge`. + */ +export const CALLEES_TRUNCATED_SENTINEL = '*'; + +/** + * Inner separator for the `BasicBlock.calleeIds` cell (resolved callee symbol + * ids). A TAB is used — NOT a space — because resolved ids embed `filePath` and + * C++ overload shape tags with multi-word primitive types (e.g. `unsigned char`, + * `long double`), so an id can legitimately contain a space; a space-joined cell + * then fragments on read and silently drops inter-procedural reach to that + * callee (#2227 tri-review). A tab cannot appear in a tree-sitter-derived id + * token (paths/identifiers/type tokens are tab-free) and round-trips intact + * through `escapeCSVField` (tab is in its preserved set) and the RFC-4180 COPY + * reader (every cell is quoted). Producer ({@link calleeIdsOfBlock}) and + * consumer (`splitCalleeIds`) import this single constant so they cannot drift. + * The sibling `callees` (leaf-name) cell stays space-joined — leaf names are + * bare identifiers and never contain a space. + */ +export const CALLEE_ID_SEP = '\t'; /** * Default per-function CFG edge cap. A pathological generated function could @@ -255,11 +285,94 @@ export const hasEmitSafeFacts = (cfg: FunctionCfg): boolean => { * no silent truncation (KTD6/R6). Block nodes are always fully emitted (their * count is bounded by the function's statement count); only edges are capped. */ +/** + * Space-joined, sorted, de-duplicated leaf callee names invoked directly in a + * block (`call`/`new` sites; the leaf of a dotted path — `child_process.exec` ⇒ + * `exec`). This is the persisted substrate for statement-precise inter-procedural + * impact: a callee reached from a function is "proven" to be impacted by a + * changed statement iff its name appears in the callees of a block in that + * statement's dependence slice. `sites` is harvested only for TS/JS under `--pdg` + * (and absent on synthetic ENTRY/EXIT), so the field is empty elsewhere and the + * bridge degrades to the prior (callgraph-equal) behavior. Space-joined because + * leaf names are identifiers (no spaces) and the field is itself one CSV cell. + */ +export function calleesOfBlock(block: BasicBlockData): string { + const names = new Set(); + for (const stmt of block.statements ?? []) { + // A statement whose recorded sites reached the per-statement cap may have + // dropped over-cap callees (the harvester stops at the cap). Flag the block + // callee-unknown so the impact bridge keeps it callgraph-equal rather than + // under-proving an absent-but-real callee. + if ((stmt.sites?.length ?? 0) >= DEFAULT_PDG_MAX_SITES_PER_STATEMENT) { + names.add(CALLEES_TRUNCATED_SENTINEL); + } + for (const site of stmt.sites ?? []) { + if (site.kind === 'member-read') continue; + const callee = site.callee; + if (!callee) continue; + const leaf = callee.slice(callee.lastIndexOf('.') + 1); + if (leaf) names.add(leaf); + } + } + return [...names].sort().join(' '); +} + +/** + * Tab-joined ({@link CALLEE_ID_SEP}), sorted, de-duplicated RESOLVED callee symbol ids invoked + * directly in a block — the SOUND parallel to {@link calleesOfBlock}'s leaf + * names (#2227 follow-up plan U3, KTD1/KTD2/KTD7). Each block site's call-site + * anchor `at` (U1) is joined by EXACT position to the per-file resolved-id map + * `fileMap` (U2's `(line,col) → Set`), so a callee reached from a + * function is proven impacted by a changed statement iff its resolved id — not + * just its leaf NAME — appears in a slice block's `calleeIds`. This eliminates + * the same-leaf-name collision (false-proven) and import-alias (false-unproven) + * the name predicate suffers on overloading languages. + * + * The site partitioning is inherited verbatim from {@link calleesOfBlock}: the + * SAME `member-read`-skip and the SAME per-statement site cap (R7) — a capped + * statement adds {@link CALLEES_TRUNCATED_SENTINEL} so the bridge keeps the + * block callee-unknown for ids too (callgraph-equal rather than under-proving). + * Because `at` is the SAME anchor the CALLS resolution keyed `atRange` on + * (KTD7), the join lands on exactly the sites the name harvest partitioned, + * including the nested-function exclusion (so a single-line inline closure's + * inner call never leaks its id into the outer block). + * + * `fileMap` is the resolved-id map for THIS file (`calleeIdAccumulator.get( + * filePath)` in run.ts). Absent (pdg off, or a file with no captured CALLS) ⇒ + * `''` — the bridge then degrades to the leaf-name fallback (R3). A site whose + * `at` is absent (pre-U1 channel) or whose position is not in the map + * contributes no id (graceful, never throws). + */ +export function calleeIdsOfBlock( + block: BasicBlockData, + fileMap: ReadonlyMap> | undefined, +): string { + if (fileMap === undefined) return ''; + const ids = new Set(); + for (const stmt of block.statements ?? []) { + // Mirror calleesOfBlock's cap signal: an over-cap statement dropped sites, + // so the id list is INCOMPLETE — flag the block callee-unknown (R7). + if ((stmt.sites?.length ?? 0) >= DEFAULT_PDG_MAX_SITES_PER_STATEMENT) { + ids.add(CALLEES_TRUNCATED_SENTINEL); + } + for (const site of stmt.sites ?? []) { + if (site.kind === 'member-read') continue; + const at = site.at; + if (!at) continue; + const resolved = fileMap.get(calleeIdPosKey(at[0], at[1])); + if (resolved === undefined) continue; + for (const id of resolved) ids.add(id); + } + } + return [...ids].sort().join(CALLEE_ID_SEP); +} + export function emitFileCfgs( graph: KnowledgeGraph, cfgs: readonly FunctionCfg[], maxEdgesPerFunction: number = DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, onWarn?: (message: string) => void, + calleeIdMap?: ReadonlyMap>, ): CfgEmitResult { const result: CfgEmitResult = { blocks: 0, edges: 0, droppedEdges: 0, cappedFunctions: 0 }; const cap = maxEdgesPerFunction > 0 ? maxEdgesPerFunction : Infinity; @@ -277,6 +390,16 @@ export function emitFileCfgs( startLine: b.startLine, endLine: b.endLine, text: b.text, + // Space-joined leaf callee names invoked in this block — the + // statement-precise inter-procedural reach substrate. Harvested from + // the per-statement `sites` (already on the side channel); dropping + // them here is what made the impact-mode bridge labeling degenerate. + callees: calleesOfBlock(b), + // Space-joined RESOLVED callee symbol ids — the SOUND parallel to + // `callees`, joined from the U2 map by each site's exact `at` + // position (#2227 follow-up U3). Absent map (pdg off / no captures) + // ⇒ `''`, and the bridge falls back to the leaf-name match (R3). + calleeIds: calleeIdsOfBlock(b, calleeIdMap), }, }); result.blocks++; @@ -361,6 +484,11 @@ export function emitFileReachingDefs( cfgs: readonly FunctionCfg[], maxEdgesPerFunction: number = DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, onWarn?: (message: string) => void, + // U12: a per-file memoized solver lets the RD-emit / harvest / taint passes + // share the SAME per-function fixpoint (this caller is its own cache bucket — + // it passes maxBlockVisits, the harvest/taint callers do not). Defaults to the + // plain solver so existing callers are unaffected. + solve: ReachingDefsSolver = computeReachingDefs, ): ReachingDefEmitResult { const result: ReachingDefEmitResult = { edges: 0, @@ -385,7 +513,7 @@ export function emitFileReachingDefs( ); continue; } - const r = computeReachingDefs(cfg, { + const r = solve(cfg, { maxFacts, maxBlockVisits: cfg.blocks.length * DEFAULT_PDG_MAX_REACHING_DEF_BLOCK_REVISITS, }); @@ -412,18 +540,47 @@ export function emitFileReachingDefs( } // Dedup to (defBlock, useBlock, binding) — facts arrive sorted, so the - // deduped order (and therefore cap truncation) is deterministic. - const seen = new Set(); - const deduped: { defBlock: number; useBlock: number; bindingIdx: number }[] = []; + // deduped order (and therefore cap truncation) is deterministic. ONE edge per + // group (the edge COUNT is unchanged — substrate/bench safe), but the FU-B-2 + // annotation AGGREGATES the FULL ordered list of (defLine, useLine) pairs for + // that group into the persisted `reason`. The first fact of a group (facts + // sort by def block, def stmt, use block, use stmt, binding) keeps the group's + // emit position; every subsequent fact of the SAME (block-pair, binding) + // appends its line pair to that group's list. Carrying the full list (not just + // the first pair) is what makes a SAME-BINDING reassignment chain recoverable: + // `acc = f(acc); acc = g(acc)` coalesces into one self-block whose + // `acc@N->acc@N+1` and `acc@N+1->acc@N+2` steps share the one group — a + // first-pair-only annotation could chain N->N+1 but never reach N+2. Dedup of + // exact-duplicate pairs within a group keeps the list compact (a `x = x + 1` + // self-fact never re-adds the same pair). + const groupIndex = new Map(); + const deduped: { + defBlock: number; + useBlock: number; + bindingIdx: number; + pairs: { defLine: number; useLine: number }[]; + }[] = []; for (const f of r.facts) { const key = `${f.def.blockIndex}:${f.use.blockIndex}:${f.bindingIdx}`; - if (seen.has(key)) continue; - seen.add(key); - deduped.push({ - defBlock: f.def.blockIndex, - useBlock: f.use.blockIndex, - bindingIdx: f.bindingIdx, - }); + const at = groupIndex.get(key); + const pair = { defLine: f.def.line, useLine: f.use.line }; + if (at === undefined) { + groupIndex.set(key, deduped.length); + deduped.push({ + defBlock: f.def.blockIndex, + useBlock: f.use.blockIndex, + bindingIdx: f.bindingIdx, + pairs: [pair], + }); + continue; + } + const list = deduped[at].pairs; + // Skip an exact-duplicate (defLine, useLine) — a self-referential statement + // (`x = x + 1`) emits the same line pair more than once; the list only needs + // each distinct step once for the projection walk. + if (!list.some((p) => p.defLine === pair.defLine && p.useLine === pair.useLine)) { + list.push(pair); + } } let emittedForFn = 0; @@ -476,7 +633,16 @@ export function emitFileReachingDefs( sourceId, targetId, confidence: 1.0, - reason: binding.name, // plain source-level name (M0/S1 verdict) — queryable + // FU-B-2: the source-level binding name (M0/S1 verdict — name FIRST so + // `pdg_query` flows stays queryable) PLUS a compact versioned annotation + // carrying the FULL ordered list of def/use source LINE pairs for this + // (block-pair, binding) group. For a self-edge (defBlock === useBlock) + // this captures the intra-block def@L→use@L' chain — including a + // SAME-BINDING reassignment chain (`acc@24->acc@25->acc@26`) — that the + // block-granular projection lost; the statement projection (pdg-impact.ts) + // walks the list forward to fixpoint to recover the coalesced block's + // interior statements. + reason: encodeReachingDefReasonPairs(binding.name, edge.pairs), }); result.edges++; emittedForFn++; diff --git a/gitnexus/src/core/ingestion/cfg/reaching-def-reason-codec.ts b/gitnexus/src/core/ingestion/cfg/reaching-def-reason-codec.ts new file mode 100644 index 000000000..d517e712b --- /dev/null +++ b/gitnexus/src/core/ingestion/cfg/reaching-def-reason-codec.ts @@ -0,0 +1,162 @@ +/** + * REACHING_DEF reason codec (PDG FU-B-2) — the ONE shared encoder/decoder for + * the source-level annotation carried on a persisted `REACHING_DEF` edge's + * `reason` column. + * + * A `REACHING_DEF` edge is `(defBlock:BasicBlock)->(useBlock:BasicBlock)` for one + * binding. The persisted columns (`from,to,type,confidence,reason,step`) are + * DEDUPED to `(defBlock, useBlock, bindingIdx)` (emit.ts), so the persisted + * edge cannot, by itself, recover the def→use chain WITHIN a coalesced + * straight-line BasicBlock: lines 7-9 of `chainCompute` collapse to one block + * and `a@7 -> b@8 -> c@9` become block-self edges with no line information. This + * codec ANNOTATES each edge with the ORDERED LIST of (defLine, useLine) source + * lines for that (block-pair, binding) group — the full set of def→use steps the + * solver produced for it — so the statement-granular intra-block chain is + * recoverable at projection time (pdg-impact.ts) WITHOUT widening the dedup key: + * the edge COUNT is unchanged (still one edge per group), only the `reason` + * carries the pair LIST (RD/taint substrate + cfg-bench budgets protected — the + * cfg-bench canon does not include the persisted `reason`). + * + * Carrying the FULL list (not just the FIRST pair) is what makes a SAME-BINDING + * reassignment chain recoverable: `acc = f(acc); acc = g(acc); acc = h(acc)` + * coalesces into one block, and ALL of `acc@24->acc@25`, `acc@25->acc@26`, + * `acc@26->acc@27` share the one `(self-block, self-block, accIdx)` group. A + * first-pair-only annotation could chain `24->25` but never reach `26`; the full + * list lets the projection walk the whole chain to fixpoint. + * + * ## Wire format (version `1`) + * + * ``` + * (legacy / pre-FU-B-2 — bare name) + * |1:: (FU-B-2, single pair) + * |1::;:;... (FU-B-2, ordered pair LIST) + * ``` + * + * The binding NAME comes FIRST, verbatim, so the established read paths keep + * working with a trivial change: `pdg_query` mode:'flows' filters the variable + * by `r.reason = $variable OR r.reason STARTS WITH $variable|` and projects the + * name via {@link decodeReachingDefReason}. Source identifiers never contain `|` + * (the structural separator), so the name is unambiguously the substring before + * the first `|`; an un-annotated reason has no `|` and decodes to itself with no + * line info. Within the annotation, `;` separates pairs and `:` separates the + * version + the two lines of each pair. ``/`` are 1-based + * decimal source lines. + * + * ## Delimiter / round-trip discipline (mirrors call-summary-codec KTD6) + * + * Every structural character (`|`, `:`, `;`, the version digit, decimal digits) + * is printable ASCII, so the encoding survives `escapeCSVField ∘ sanitizeUTF8` + * (csv-generator.ts) byte-exact. The decoder NEVER throws — anything not a + * well-formed version-`1` annotation degrades to "name only, no pairs" (the sound + * default: the projection then falls back to block-start granularity exactly as + * before FU-B-2). A malformed individual pair within an otherwise-well-formed + * list is dropped; the well-formed pairs are kept. + */ + +/** One-character format version prefix. Bump on any wire-format change. */ +export const REACHING_DEF_REASON_CODEC_VERSION = '1'; + +/** + * Structural separator between the binding name and the versioned annotation. + * Source identifiers cannot contain it, so the name is the substring before the + * first occurrence (and a name with no occurrence is a legacy bare-name reason). + */ +const NAME_SEP = '|'; + +/** Separator between consecutive (defLine:useLine) pairs in the annotation. */ +const PAIR_SEP = ';'; + +/** One def→use source-line step within a coalesced block's self chain. */ +export interface DefUseLinePair { + /** 1-based def source line. */ + readonly defLine: number; + /** 1-based use source line. */ + readonly useLine: number; +} + +/** A decoded REACHING_DEF reason. `pairs` is empty for a legacy (un-annotated) + * reason. `defLine`/`useLine` mirror the FIRST pair for back-compat consumers. */ +export interface DecodedReachingDefReason { + /** The source-level binding name (always present — the legacy payload). */ + readonly name: string; + /** + * The ordered list of (defLine, useLine) steps the FU-B-2 annotation carries. + * Empty for a legacy / malformed / un-annotated reason. + */ + readonly pairs: readonly DefUseLinePair[]; + /** 1-based def source line of the FIRST pair (back-compat; absent if none). */ + readonly defLine?: number; + /** 1-based use source line of the FIRST pair (back-compat; absent if none). */ + readonly useLine?: number; +} + +/** Whether a (defLine, useLine) pair is a well-formed 1-based-or-0 integer pair. */ +function isValidPair(defLine: number, useLine: number): boolean { + return Number.isInteger(defLine) && Number.isInteger(useLine) && defLine >= 0 && useLine >= 0; +} + +/** + * Encode a binding name + its ordered (defLine, useLine) step list into the + * versioned `reason` wire string. Deterministic; never throws. Malformed / + * negative pairs are dropped (defensive — the solver always passes 1-based + * integers); if NO valid pair survives the result degrades to the bare name + * (legacy form) so a malformed annotation never fabricates bad lines. The name + * is written verbatim FIRST (see the module doc), so a name that — + * pathologically — already contains `|` would be re-decoded with a truncated + * name; binding names are source identifiers, which never contain `|`, so this + * cannot occur for real input (and would only lose line precision, never corrupt + * the substrate). + */ +export function encodeReachingDefReasonPairs( + name: string, + pairs: ReadonlyArray, +): string { + const valid = pairs.filter((p) => isValidPair(p.defLine, p.useLine)); + if (valid.length === 0) return name; + const body = valid.map((p) => `${p.defLine}:${p.useLine}`).join(PAIR_SEP); + return `${name}${NAME_SEP}${REACHING_DEF_REASON_CODEC_VERSION}:${body}`; +} + +/** + * Single-pair convenience over {@link encodeReachingDefReasonPairs} — kept for + * call sites and tests that carry exactly one def→use step. + */ +export function encodeReachingDefReason(name: string, defLine: number, useLine: number): string { + return encodeReachingDefReasonPairs(name, [{ defLine, useLine }]); +} + +/** + * Decode a REACHING_DEF `reason` wire string into its binding name + (when + * present) the FU-B-2 def/use source-line pair LIST. Never throws — a non-string, + * an un-annotated bare name, or a malformed annotation all yield `{ name, pairs: + * [] }` (the sound default: the consumer falls back to block-start granularity). + * A well-formed `|1::;:;...` yields every well-formed pair + * (a single malformed pair is dropped, the rest kept); `defLine`/`useLine` mirror + * the first pair for back-compat consumers. + */ +export function decodeReachingDefReason(reason: unknown): DecodedReachingDefReason { + const raw = typeof reason === 'string' ? reason : ''; + const sep = raw.indexOf(NAME_SEP); + if (sep === -1) return { name: raw, pairs: [] }; + const name = raw.slice(0, sep); + const annotation = raw.slice(sep + 1); + // Annotation is `::;:;...`. Split off the version + // prefix once: the first colon ends the version token; the remainder is the + // `;`-separated pair body. + const firstColon = annotation.indexOf(':'); + if (firstColon === -1 || annotation.slice(0, firstColon) !== REACHING_DEF_REASON_CODEC_VERSION) { + return { name, pairs: [] }; + } + const body = annotation.slice(firstColon + 1); + const pairs: DefUseLinePair[] = []; + for (const chunk of body.split(PAIR_SEP)) { + const parts = chunk.split(':'); + if (parts.length !== 2) continue; // malformed pair — drop it, keep the rest + const defLine = Number(parts[0]); + const useLine = Number(parts[1]); + if (!isValidPair(defLine, useLine)) continue; + pairs.push({ defLine, useLine }); + } + if (pairs.length === 0) return { name, pairs: [] }; + return { name, pairs, defLine: pairs[0].defLine, useLine: pairs[0].useLine }; +} diff --git a/gitnexus/src/core/ingestion/cfg/reaching-defs.ts b/gitnexus/src/core/ingestion/cfg/reaching-defs.ts index 4e9f3c23b..ef36a8ed8 100644 --- a/gitnexus/src/core/ingestion/cfg/reaching-defs.ts +++ b/gitnexus/src/core/ingestion/cfg/reaching-defs.ts @@ -243,6 +243,37 @@ export function computeReachingDefs(cfg: FunctionCfg, limits?: ReachingDefsLimit return solveReachingDefs(cfg, limits, computeInSetsAuto); } +/** A reaching-defs solver — {@link computeReachingDefs} or a memoized wrapper. */ +export type ReachingDefsSolver = (cfg: FunctionCfg, limits?: ReachingDefsLimits) => FunctionDefUse; + +/** + * Per-file memoized reaching-defs solver (#2227 tri-review, U12). Under `--pdg` + * the SAME per-function RD fixpoint was solved 3–4× per analyze (RD emit + + * call-summary harvest + taint + summary harvest). Cache by (cfg identity, + * limits) so each DISTINCT solve runs once: the RD-emit bucket (passes + * `maxBlockVisits`) and the harvest/taint bucket (does not) stay byte-identical + * to their inline solves because the limits are part of the key. Lazy — solves + * on first request, so the taint zero-match fast path still skips its solve. + * Create one per FILE and drop it after the file to bound the per-function + * `facts` arrays (100k+ objects on a huge function) from going whole-repo. + */ +export function createMemoizedReachingDefs(): ReachingDefsSolver { + const cache = new Map>(); + return (cfg, limits) => { + const key = `${limits?.maxFacts ?? ''}|${limits?.maxBlockVisits ?? ''}`; + let byKey = cache.get(cfg); + if (byKey === undefined) { + byKey = new Map(); + cache.set(cfg, byKey); + } + const hit = byKey.get(key); + if (hit !== undefined) return hit; + const result = computeReachingDefs(cfg, limits); + byKey.set(key, result); + return result; + }; +} + /** * Dense GEN/KILL monotone worklist — the original (#2082 M2) reaching-defs * solver. As of #2201 it plays two roles: (1) the production dispatcher diff --git a/gitnexus/src/core/ingestion/cfg/types.ts b/gitnexus/src/core/ingestion/cfg/types.ts index 53988e687..55f95923a 100644 --- a/gitnexus/src/core/ingestion/cfg/types.ts +++ b/gitnexus/src/core/ingestion/cfg/types.ts @@ -40,6 +40,20 @@ export interface BindingEntry { * `name@module` in edge ids instead of `name:line:col`. */ readonly synthetic?: boolean; + /** + * For `kind: 'param'` bindings only: the 0-based ENCLOSING TOP-LEVEL FORMAL + * position this binding belongs to — the index a call site's argument position + * joins against (PDG FU-C). For a simple identifier formal this equals the + * param's ordinal; for a DESTRUCTURED/REST formal every inner name carries the + * SAME formal index (`function f({a, b}, c)` ⇒ a:0, b:0, c:1), so a downstream + * positional consumer never mistakes the destructured-object formal for a later + * simple formal. Set by the per-language `declareParams`; OMITTED when the + * producer does not (yet) supply it — a consumer that needs a sound formal + * position MUST treat a param binding without `formalIndex` as unknown and fall + * back conservatively (never attribute a flattened ordinal to a formal slot). + * Omit-when-absent (pre-upgrade durable channels stay valid; JSON-plain). + */ + readonly formalIndex?: number; } /** @@ -126,6 +140,35 @@ export interface SiteRecord { * included; dynamic `req[key]` is never recorded — documented KTD10 FN). */ readonly property?: string; + /** + * Call-site anchor source position `[line (1-based), column (0-based)]` for + * call/new sites only — member-read sites omit it (the resolved-id join only + * consumes call/new). Recorded by the harvester at the call/new node where it + * reads the callee, so the later resolved-callee-id join inherits this + * harvester's exact (nested-function-excluded — see line 150) site + * partitioning (#2227 follow-up plan KTD1). + * + * ANCHOR ALIGNMENT (plan KTD7 — load-bearing): this MUST be the SAME position + * the CALLS-edge resolution keys its `atRange` on, because a downstream unit + * joins the two by EXACT position. That anchor is the WHOLE call/new + * expression node's start — `nodeToCapture('@reference.call.*', node)` in the + * scope-extractor anchors `@reference.call.free/.member/.constructor` on the + * `call_expression`/`new_expression` (TS) / `method_invocation`/ + * `object_creation_expression` (Java) node itself (the callee identifier / + * member property is the `@reference.name` SUB-tag, never the anchor — see + * `anchorCaptureFor` + `KNOWN_SUB_TAGS` in scope-extractor.ts, and + * `atRange: anchor.range` at scope-extractor.ts:1030). So for a bare call + * `foo(x)`, a member call `arr.map(x)`, and a namespaced/chained call + * `a.b.c(x)` alike, `at` is the start of the enclosing call/new expression + * node — the harvester's `visitCall`/`visitNew` receives exactly that node and + * records `[node.startPosition.row + 1, node.startPosition.column]`. (For a + * member call the call expression starts at the receiver, e.g. `arr` in + * `arr.map(x)`, and the CALLS anchor starts there too — they match.) + * + * Omit-when-absent (pre-upgrade durable channels stay valid; JSON-plain; NOT + * named `nodeId` per the reviver hazard above). + */ + readonly at?: readonly [number, number]; } /** diff --git a/gitnexus/src/core/ingestion/cfg/visitors/c-cpp-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/c-cpp-harvest.ts index 730bb8930..13e259731 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/c-cpp-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/c-cpp-harvest.ts @@ -459,7 +459,9 @@ export class CCppHarvester extends ScopeTreeHarvester { private visitCall(node: SyntaxNode, acc: FactAccumulator, kind: 'call' | 'new'): void { const calleeNode = node.childForFieldName(kind === 'new' ? 'type' : 'function'); const argsNode = node.childForFieldName('arguments'); - const siteIdx = acc.openCallSite(kind); + // `node` IS the call/new expression — the SAME node the scope-extractor + // anchors `@reference.call.*` (its `atRange`) on (KTD7). + const siteIdx = acc.openCallSite(kind, [node.startPosition.row + 1, node.startPosition.column]); acc.pushFrame(siteIdx); let calleePath: string | undefined; if (calleeNode) { diff --git a/gitnexus/src/core/ingestion/cfg/visitors/call-site-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/call-site-harvest.ts index ed85bc2dc..9ba67ef36 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/call-site-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/call-site-harvest.ts @@ -29,6 +29,7 @@ * NOTE: nothing serialized here may carry a field named `nodeId` — the durable * parsedfile-store reviver dedups objects keyed on that field name. */ +import type { SyntaxNode } from '../../utils/ast-helpers.js'; import type { SiteArgOccurrence, SiteRecord, StatementFacts } from '../types.js'; /** Mutable build-time view of a {@link SiteRecord}. */ @@ -44,6 +45,8 @@ interface MutableSite { requireArg?: string; object?: number; property?: string; + /** Call-site anchor position — see {@link SiteRecord.at}. Call/new only. */ + at?: [number, number]; } /** @@ -214,8 +217,14 @@ export class CallSiteFactAccumulator { * Returns the new site index, or -1 when the per-statement site cap is hit * (the caller threads -1 through `pushFrame`/`setSite*`, all of which no-op on * a sentinel index — see {@link DEFAULT_PDG_MAX_SITES_PER_STATEMENT}). + * + * `at` is the call/new node's anchor position `[line (1-based), col (0-based)]` + * — the SAME position the CALLS-edge resolution keys on (see + * {@link SiteRecord.at} for the KTD7 alignment); the harvester passes its + * `visitCall`/`visitNew` node's `startPosition` so the downstream resolved-id + * join lands by exact position. */ - openCallSite(kind: 'call' | 'new'): number { + openCallSite(kind: 'call' | 'new', at?: readonly [number, number]): number { if (this.sites.length >= DEFAULT_PDG_MAX_SITES_PER_STATEMENT) { this._sitesTruncated = true; return -1; @@ -223,6 +232,7 @@ export class CallSiteFactAccumulator { const site: MutableSite = { kind }; const parent = this.innermostArgPosition(); if (parent) site.parent = parent; + if (at) site.at = [at[0], at[1]]; this.sites.push(site); return this.sites.length - 1; } @@ -372,3 +382,64 @@ const finalizeSite = (site: MutableSite): SiteRecord => { } return site as SiteRecord; }; + +/** + * Per-grammar hooks the shared {@link finalizeChain} terminal needs but cannot + * name itself (it carries no tree-sitter literals — see the file header). Each + * harvester supplies the two callbacks bound to its own `this`. + */ +export interface ChainTerminalHooks { + /** Resolve a binding-target node to its function-table binding index. */ + resolve(node: SyntaxNode): number; + /** + * Walk a NON-identifier chain root for its uses + nested sites (the terminal's + * `else` branch — `self.x.f()`, `foo().bar`, a tuple index, etc.). + */ + walkRoot(node: SyntaxNode): void; +} + +/** + * Shared `walkChain` TERMINAL (#2227 follow-up, plan KTD5/U8) — the byte-identical + * post-unwind block the Go / Kotlin / Swift / Rust / Python harvesters all ran + * after walking their grammar-specific access chain (`selector_expression` / + * `navigation_expression` / `field_expression` / `attribute`) into an + * `accesses: string[]` list and a resolved root node `cur`. + * + * It records the chain-root identifier as a use, emits at most ONE member-read + * site — the INNERMOST access — when the root is an identifier (suppressed by + * `skipFinalRead` when that access IS the callee, carried by the dotted path + * instead), and builds the dotted path `[root, ...accesses].join('.')`. The only + * per-grammar bit is the root identifier node type, supplied via `isRootIdType` + * (`'identifier'` for Go/Rust/Python, `'simple_identifier'` for Kotlin/Swift); + * the `resolve` / `walkRoot` callbacks bind the harvester's own methods. The + * `addUse` / `addMemberRead` machinery is on the accumulator itself, so it is + * called directly (no callback). Behavior is identical to the inlined terminals + * this replaces — the per-language harvest tests are the characterization lock. + */ +export function finalizeChain( + acc: CallSiteFactAccumulator, + cur: SyntaxNode, + accesses: readonly string[], + skipFinalRead: boolean, + isRootIdType: (type: string) => boolean, + hooks: ChainTerminalHooks, +): { path?: string; rootIdx?: number } { + let rootIdx: number | undefined; + let rootSegment: string | undefined; + if (isRootIdType(cur.type) && cur.text !== '_') { + rootIdx = hooks.resolve(cur); + acc.addUse(rootIdx); + rootSegment = cur.text; + } else { + hooks.walkRoot(cur); + } + const innermost = accesses[0]; + if (rootIdx !== undefined && innermost && !(skipFinalRead && accesses.length === 1)) { + acc.addMemberRead(rootIdx, innermost); + } + const path = + rootSegment !== undefined && accesses.every((a) => a !== '') + ? [rootSegment, ...accesses].join('.') + : undefined; + return { path, rootIdx }; +} diff --git a/gitnexus/src/core/ingestion/cfg/visitors/csharp-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/csharp-harvest.ts index bcc615ab2..f32292527 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/csharp-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/csharp-harvest.ts @@ -466,7 +466,9 @@ export class CsharpHarvester extends ScopeTreeHarvester { private visitCall(node: SyntaxNode, acc: FactAccumulator, kind: 'call' | 'new'): void { const calleeNode = node.childForFieldName(kind === 'new' ? 'type' : 'function'); const argsNode = node.childForFieldName('arguments'); - const siteIdx = acc.openCallSite(kind); + // `node` IS the call/object-creation expression — the SAME node the + // scope-extractor anchors `@reference.call.*` (its `atRange`) on (KTD7). + const siteIdx = acc.openCallSite(kind, [node.startPosition.row + 1, node.startPosition.column]); acc.pushFrame(siteIdx); let calleePath: string | undefined; if (calleeNode) { diff --git a/gitnexus/src/core/ingestion/cfg/visitors/dart-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/dart-harvest.ts index 9234c9dd0..aab133ef7 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/dart-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/dart-harvest.ts @@ -1,11 +1,46 @@ /** * Dart def/use harvester (#2195) — the Dart analogue of - * {@link import('./kotlin-harvest.js').KotlinHarvester} and the Swift / Python / - * Rust harvesters. Like them it harvests NO call-site `sites[]` (the call-site - * taint substrate is a later step): it emits only the per-function binding table - * ({@link BindingEntry}[]) plus {@link StatementFacts} (defs / uses / mayDefs) via - * a local {@link FactAccumulator} with no site machinery, so the produced facts - * never carry a `sites` key. + * {@link import('./python-harvest.js').PythonHarvester} and the C-family + * harvesters. Like Python it harvests per-function binding tables + * ({@link BindingEntry}[]) plus {@link StatementFacts} (defs / uses / mayDefs) + * AND a taint {@link import('../types.js').SiteRecord} per call / `new` (callee + * path, receiver, per-arg occurrence entries, result defs, spread marker, and an + * `at` anchor) via the shared {@link CallSiteFactAccumulator} — the same site + * substrate the C-family / Go / TS / Python harvesters emit. + * + * DART HAS NO `call_expression` NODE (verified by a real parse — see below). A + * call is a FLAT SIBLING RUN under a container (`expression_statement`, + * `argument`, an `initialized_variable_definition`'s `value` field, + * `await_expression`, …): a chain HEAD (`identifier` / `this` / `super` / a + * parenthesized expr) immediately followed by one or more `selector` siblings. + * A `selector` whose inner is an `argument_part` is the CALL marker (the prefix + * up to it is the callee); a `selector` whose inner is an + * `unconditional_assignable_selector` / `conditional_assignable_selector` + * (`.name` / `?.name`) is a member access. So `foo(a, b)` parses as + * `identifier foo` + `selector (a, b)`; `obj.method(x)` as `identifier obj` + + * `selector .method` + `selector (x)`; `a.b.c()` as `a` + `.b` + `.c` + `()`. + * A `new Foo(…)` IS a single `new_expression` node (`type_identifier` + + * `arguments`) — the only `kind: 'new'` shape. An UpperCamelCase bare call + * `Foo(…)` is an IMPLICIT constructor by Dart convention but is structurally a + * free call (`identifier` + `selector(argument_part)`), so it stays + * `kind: 'call'` (matching the scope-extractor, which tags it + * `@reference.call.constructor` on the same callee identifier — see below). + * + * ANCHOR ALIGNMENT (plan KTD7 — load-bearing): a call site's `at` MUST be the + * SAME `[line (1-based), col (0-based)]` the Dart CALLS resolution keys its + * `atRange` on, because a downstream unit joins the two by EXACT position. Dart + * has no whole-call node, so the scope-extractor anchors the CALLS reference NOT + * on a call expression but on the callee NAME identifier + * (`captures.ts emitSelectorReference`): + * - a FREE / implicit-constructor call `foo(…)` / `Foo(…)` → + * `@reference.call.free` / `.constructor` anchored on the callee `identifier` + * (`prev`), so `at` = that identifier's start. + * - a MEMBER call `obj.method(…)` → `@reference.call.member` anchored on the + * method-name `identifier` (`nameId`, inside the `.method` selector), so + * `at` = the method-name identifier's start — NOT the receiver `obj`. + * (A `new_expression` is NOT captured for CALLS by the Dart scope-resolution + * today, so a `new` site's `at` simply finds no resolved id — graceful, never a + * mis-join. A cascade `a..m(…)` resolves as a FREE call on its method name.) * * Runs in the parse worker next to the Dart CFG visitor. Output is the binding * table the {@link import('../cfg-builder.js').CfgBuilder} stamps onto the CFG, @@ -78,7 +113,13 @@ */ import type { SyntaxNode } from '../../utils/ast-helpers.js'; import type { BindingEntry, StatementFacts } from '../types.js'; -import { DefUseAccumulator as FactAccumulator } from './call-site-harvest.js'; +import { CallSiteFactAccumulator as FactAccumulator } from './call-site-harvest.js'; + +/** Selector inners that are a `.name` / `?.name` member access (not a call). */ +const ASSIGNABLE_SELECTOR_TYPES = new Set([ + 'unconditional_assignable_selector', + 'conditional_assignable_selector', +]); /** Node types that own a nested CFG — their subtrees are opaque to harvesting. */ const NESTED_FUNCTION_TYPES = new Set(['function_expression', 'function_body']); @@ -95,6 +136,14 @@ export class DartHarvester { private readonly fnId: number; /** >0 while walking a conditionally-evaluated subexpression — defs become may-defs. */ private conditionalDepth = 0; + /** + * Chain-head / `new_expression` node id → binding indices its single-target + * result is assigned to (`var x = f()` / `x = g()` ⇒ `[x]`). Populated just + * before the value walk reaches the call (see {@link registerResultDefs}) and + * consumed by {@link visitChainCall} / {@link visitNew}. Mirrors the Python / + * Go harvesters' `resultDefTargets`. + */ + private readonly resultDefTargets = new Map(); /** * @param fnNode The function-bearing node: a `function_body` (whose previous @@ -394,9 +443,16 @@ export class DartHarvester { this.use(node, acc); return; case 'initialized_variable_definition': { - const value = node.childForFieldName('value'); - if (value) this.walkValue(value, acc); const name = node.childForFieldName('name'); + // The `value` field can REPEAT across a Dart postfix run (`= g` `(y)`): + // collect every `value`-tagged child as one chain and walk them together + // so a member/free call across the run is harvested as one site. + const valueRun = this.fieldRun(node, 'value'); + // Register result-defs BEFORE the value walk so the call the value walk + // reaches carries them — single identifier target only (`var x = f()`); + // a trailing comma-separated `b = 2` declarator attaches nothing. + if (name && valueRun.length > 0) this.registerRunResultDefs(valueRun, [name]); + if (valueRun.length > 0) this.walkRun(valueRun, acc); if (name) this.def(name, acc); // Trailing comma-separated bindings (`var a = 1, b = 2;`): each // `initialized_identifier` is an `identifier` + its own value expr. @@ -413,18 +469,18 @@ export class DartHarvester { case 'assignment_expression': { const lvalue = node.childForFieldName('left'); const op = node.childForFieldName('operator'); - const value = node.childForFieldName('right'); - if (value) this.walkValue(value, acc); - // A `right` field can repeat (an identifier + trailing selectors): walk - // every named child after the operator that isn't the lvalue. - for (const c of node.namedChildren) { - if (c === lvalue) continue; - if (c.type === 'assignable_expression') continue; - if (c === value) continue; - this.walkValue(c, acc); + // The `right` field can REPEAT across a postfix run (`= obj` `.m` `(y)`): + // collect every `right`-tagged child and group the run so a call across + // it is one site. + const rhs = this.fieldRun(node, 'right'); + const scalar = lvalue ? this.scalarAssignTarget(lvalue) : undefined; + // A plain `x = ` attaches `resultDefs: [x]` (compound `+=` does not — + // the prior value flows in too). + if (scalar && op?.text === '=' && rhs.length > 0) { + this.registerRunResultDefs(rhs, [scalar]); } + if (rhs.length > 0) this.walkRun(rhs, acc); if (lvalue) { - const scalar = this.scalarAssignTarget(lvalue); if (scalar) { this.def(scalar, acc); if (op && op.text !== '=') this.use(scalar, acc); // compound assign reads too @@ -466,8 +522,10 @@ export class DartHarvester { return; } case 'selector': { - // `.name` / `(...args)` — a member-access suffix name is not a scalar - // binding; walk the argument part for uses but skip the bare property id. + // FALLBACK: a `selector` reached OUTSIDE a postfix run (normal chains are + // grouped + harvested by `walkChildren`/`walkRun`). `.name` / `(...args)` + // — a member-access suffix name is not a scalar binding; walk the argument + // part for uses but skip the bare property id. No site is emitted here. for (const c of node.namedChildren) { if ( c.type === 'unconditional_assignable_selector' || @@ -480,36 +538,20 @@ export class DartHarvester { return; } case 'logical_and_expression': - case 'logical_or_expression': { - // `a && b` / `a || b` — the right operand is conditionally evaluated. - const operands = node.namedChildren.filter((c) => !COMMENT_TYPES.has(c.type)); - if (operands.length > 0) this.walkValue(operands[0], acc); - for (let i = 1; i < operands.length; i++) { - const rhs = operands[i]; - this.conditional(() => this.walkValue(rhs, acc)); - } + case 'logical_or_expression': + // `a && b` / `a || b` — the right operand is conditionally evaluated. The + // operand may be a flattened postfix run (`a && g(x)` ⇒ `a`, `g`, + // `selector`), so group runs and demote everything after the operator. + this.walkBinaryConditional(node, acc, new Set(['&&', '||'])); return; - } - case 'if_null_expression': { + case 'if_null_expression': // `a ?? b` — the right operand only evaluates when the left is null. - const operands = node.namedChildren.filter((c) => !COMMENT_TYPES.has(c.type)); - if (operands.length > 0) this.walkValue(operands[0], acc); - for (let i = 1; i < operands.length; i++) { - const rhs = operands[i]; - this.conditional(() => this.walkValue(rhs, acc)); - } + this.walkBinaryConditional(node, acc, new Set(['??'])); return; - } - case 'conditional_expression': { + case 'conditional_expression': // `c ? a : b` — the condition runs always; both arms are conditional. - const operands = node.namedChildren.filter((c) => !COMMENT_TYPES.has(c.type)); - if (operands.length > 0) this.walkValue(operands[0], acc); - for (let i = 1; i < operands.length; i++) { - const arm = operands[i]; - this.conditional(() => this.walkValue(arm, acc)); - } + this.walkBinaryConditional(node, acc, new Set(['?', ':'])); return; - } case 'switch_expression': { // `switch (x) { p1 => a, p2 => b }` (Dart 3): the subject runs always; // each arm (pattern + value) is conditional, so a def inside an arm value @@ -524,6 +566,11 @@ export class DartHarvester { } return; } + case 'new_expression': + // `new Foo(args)` — the only single-node call shape Dart has. Constructor + // site (`kind: 'new'`). + this.visitNew(node, acc); + return; case 'inferred_type': case 'final_builtin': case 'type_identifier': @@ -531,13 +578,380 @@ export class DartHarvester { // Binding keyword / type position — no scalar value uses. return; default: - for (let i = 0; i < node.namedChildCount; i++) { - const c = node.namedChild(i); - if (c) this.walkValue(c, acc); - } + // A container (`expression_statement`, `argument`, `await_expression`, + // `cascade_section`'s parent, …) whose children may form Dart postfix + // call/access RUNS (`identifier` + `selector*`). Group runs so a call is + // harvested as one site; non-run children walk normally. + this.walkChildren(node.namedChildren, acc); } } + // ── Dart postfix call/access chains (#2227 follow-up) ───────────────────── + + /** + * Walk a container's named children, coalescing each Dart postfix RUN — a + * chain HEAD immediately followed by one or more `selector` (and/or a + * `cascade_section`) siblings — into a single {@link walkRun} so a member / + * free call across the run is harvested as ONE call site. A child that does + * not start a run walks via {@link walkValue} as before. + */ + private walkChildren(children: readonly SyntaxNode[], acc: FactAccumulator): void { + const named = children.filter((c) => !COMMENT_TYPES.has(c.type)); + let i = 0; + while (i < named.length) { + const head = named[i]; + // A run continues over immediately-following `selector` / `cascade_section` + // siblings (the postfix suffixes applied to `head`). + let end = i + 1; + while (end < named.length && this.isSuffix(named[end])) end++; + if (end > i + 1) { + this.walkRun(named.slice(i, end), acc); + } else { + this.walkValue(head, acc); + } + i = end; + } + } + + /** A postfix-run suffix node: a `.name`/`(args)` `selector` or a `..m()` cascade. */ + private isSuffix(node: SyntaxNode): boolean { + return node.type === 'selector' || node.type === 'cascade_section'; + } + + /** + * Walk a binary / ternary expression whose operands are FLATTENED across the + * node's children (`a ?? g(x)` ⇒ children `a`, `??`, `g`, `selector`). The + * children before the FIRST boundary operator (`boundaries`) run + * unconditionally; everything after a boundary is conditionally evaluated (a + * may-def context). Each segment is grouped via {@link walkChildren} so a + * postfix call split across children (`g` + `selector`) is one site. + */ + private walkBinaryConditional( + node: SyntaxNode, + acc: FactAccumulator, + boundaries: ReadonlySet, + ): void { + const segments: SyntaxNode[][] = [[]]; + for (let i = 0; i < node.childCount; i++) { + const c = node.child(i); + if (!c || COMMENT_TYPES.has(c.type)) continue; + if (!c.isNamed && boundaries.has(c.text)) { + segments.push([]); + continue; + } + if (c.isNamed) segments[segments.length - 1].push(c); + } + // First segment unconditional; the rest are conditional arms / right operands. + if (segments[0].length > 0) this.walkChildren(segments[0], acc); + for (let i = 1; i < segments.length; i++) { + const seg = segments[i]; + if (seg.length > 0) this.conditional(() => this.walkChildren(seg, acc)); + } + } + + /** + * Walk one postfix run `[head, suffix*]`. A lone node (no suffixes) is just a + * value walk. A run with suffixes is a Dart call/access chain: each `selector` + * whose inner is an `argument_part` is a call applied to the prefix; an + * assignable `.name`/`?.name` selector is a member access; a `cascade_section` + * with an `argument_part` is a free call on its method name. + */ + private walkRun(run: readonly SyntaxNode[], acc: FactAccumulator): void { + if (run.length === 1) { + this.walkValue(run[0], acc); + return; + } + const head = run[0]; + const suffixes = run.slice(1); + const hasCascade = suffixes.some((s) => s.type === 'cascade_section'); + // A cascade target (`a..m()`) is read once; its cascade calls are FREE calls + // on the method name (matching the scope-extractor's cascade classification). + if (hasCascade) { + this.walkValue(head, acc); + for (const s of suffixes) { + if (s.type === 'cascade_section') this.visitCascade(s, acc); + else this.walkValue(s, acc); + } + return; + } + // Plain postfix chain: walk left→right, opening a call site at each + // `selector(argument_part)`. The callee path / receiver come from the prefix. + // Only the LAST call in the run receives the binding result (`var x = + // a.b().c()` ⇒ x is `.c()`'s result, not `.b()`'s). + const lastCallIdx = this.lastCallSelectorIndex(run); + let chainStartUseRecorded = false; + let rootIdx: number | undefined; + let rootSegment: string | undefined; + const accesses: string[] = []; // dotted member segments since the last call + for (let i = 0; i < run.length; i++) { + const node = run[i]; + if (node === head) { + const resolvedHead = this.chainHead(head); + if (resolvedHead) { + rootIdx = this.resolve(resolvedHead); + rootSegment = resolvedHead.text; + } else { + // A non-identifier head (parenthesized expr, literal) — walk it for + // uses; it has no static root segment. + this.walkValue(head, acc); + } + continue; + } + // node is a `selector`. + const inner = node.namedChild(0); + if (inner?.type === 'argument_part') { + // Call marker — `prefix(args)`. Record the chain root use (once). For a + // FREE call the head IS the callee NAME — a statement-level use but NOT a + // value occurrence in any enclosing argument (`exec(escape(x))` must not + // put `escape` into exec's arg 0). For a MEMBER call the head is the + // RECEIVER (a real value that launders taint) — a normal occurrence use. + if (rootIdx !== undefined && !chainStartUseRecorded) { + if (accesses.length === 0) acc.addUseWithoutOccurrence(rootIdx); + else acc.addUse(rootIdx); + chainStartUseRecorded = true; + } + // The accesses since the last call form the callee tail; the FINAL access + // IS the callee (carried by the path, `skipFinalRead`), but an inner + // access is a member READ — `a.b.c()` reads `a.b` then calls `.c` (mirror + // the Go / Python `walkChain` innermost member-read). A single access + // (`obj.method()`) is just the callee — no read. + if (rootIdx !== undefined && accesses.length >= 2) { + acc.addMemberRead(rootIdx, accesses[0]); + } + const anchor = this.callAnchor(run, i, head); + this.visitChainCall(node, inner, acc, { + rootIdx, + callee: this.calleePath(rootSegment, accesses), + anchor, + isMember: accesses.length > 0, + isLastCall: i === lastCallIdx, + }); + // After a call the result is opaque — subsequent accesses have no static + // root (the call return value), so drop the path. + accesses.length = 0; + rootSegment = undefined; + rootIdx = undefined; + continue; + } + if (inner && ASSIGNABLE_SELECTOR_TYPES.has(inner.type)) { + // `.name` / `?.name` member access — extends the dotted path. A trailing + // access NOT followed by a call is a value-position member READ (record + // the root use + at most one member-read site at the innermost access). + const seg = this.selectorName(inner); + if (seg) accesses.push(seg.text); + const isLastAndNotCall = i === run.length - 1; + if (isLastAndNotCall && rootIdx !== undefined) { + if (!chainStartUseRecorded) { + acc.addUse(rootIdx); + chainStartUseRecorded = true; + } + // Innermost access is a member read (`a.b` in `a.b.c` value position). + if (accesses.length >= 1) acc.addMemberRead(rootIdx, accesses[0]); + } + continue; + } + // An index `[i]` / other selector — walk its inner for uses. + this.walkValue(node, acc); + } + // A chain that ended without any call but had a root (`a.b` as a whole value + // read) still records its root use even when the access loop didn't (e.g. a + // single non-call selector run reached here without a call). + if (rootIdx !== undefined && !chainStartUseRecorded) acc.addUse(rootIdx); + } + + /** The chain root binding node — an `identifier` head (not `this`/`super`/literal). */ + private chainHead(head: SyntaxNode): SyntaxNode | undefined { + return head.type === 'identifier' && head.text !== '_' ? head : undefined; + } + + /** The bound name identifier of an assignable selector inner (`.name` ⇒ `name`). */ + private selectorName(inner: SyntaxNode): SyntaxNode | undefined { + for (let i = inner.namedChildCount - 1; i >= 0; i--) { + const c = inner.namedChild(i); + if (c?.type === 'identifier') return c; + } + return undefined; + } + + /** Dotted callee path `root.a.b` (or undefined when the root is not an identifier). */ + private calleePath( + rootSegment: string | undefined, + accesses: readonly string[], + ): string | undefined { + if (rootSegment === undefined) return undefined; + return [rootSegment, ...accesses].join('.'); + } + + /** + * The `at` anchor for a call `selector` at `run[i]`, byte-aligned with the + * Dart CALLS `atRange` (see file header): a FREE / implicit-constructor call + * (the call selector immediately follows the chain head) anchors on the head + * identifier; a MEMBER call anchors on the method-NAME identifier of the + * preceding `.method` selector. + */ + private callAnchor(run: readonly SyntaxNode[], i: number, head: SyntaxNode): [number, number] { + const prev = run[i - 1]; + if (prev && prev.type === 'selector') { + const inner = prev.namedChild(0); + const nameId = inner ? this.selectorName(inner) : undefined; + if (nameId) return [nameId.startPosition.row + 1, nameId.startPosition.column]; + } + // Free / constructor call — anchor on the chain head (the callee identifier). + return [head.startPosition.row + 1, head.startPosition.column]; + } + + /** + * Open + populate a call site for a Dart postfix `prefix(args)` call. The + * callee NAME is a statement-level use (recorded as the chain root above, or + * here for a bare free call), NOT a value occurrence in any enclosing argument. + */ + private visitChainCall( + selector: SyntaxNode, + argPart: SyntaxNode, + acc: FactAccumulator, + info: { + rootIdx: number | undefined; + callee: string | undefined; + anchor: [number, number]; + isMember: boolean; + isLastCall: boolean; + }, + ): void { + const siteIdx = acc.openCallSite('call', info.anchor); + acc.pushFrame(siteIdx); + if (info.callee !== undefined) acc.setSiteCallee(siteIdx, info.callee); + // A member call (`obj.m(x)`, `a.b.c(x)`) launders taint through its receiver + // root; a bare free call (`foo(x)`) has no receiver. + if (info.isMember && info.rootIdx !== undefined) acc.setSiteReceiver(siteIdx, info.rootIdx); + // Only the run's terminal call receives the binding result (its `.parent` is + // the run's shared `initialized_variable_definition` / `assignment_expression`). + if (info.isLastCall) { + const resultDefs = this.resultDefTargets.get(selector.parent?.id ?? -1); + if (resultDefs !== undefined) acc.setSiteResultDefs(siteIdx, resultDefs); + } + this.walkArguments(argPart, siteIdx, acc); + acc.popFrame(); + } + + /** Index in `run` of the LAST call-marker selector (`selector(argument_part)`). */ + private lastCallSelectorIndex(run: readonly SyntaxNode[]): number { + for (let i = run.length - 1; i >= 0; i--) { + const n = run[i]; + if (n.type === 'selector' && n.namedChild(0)?.type === 'argument_part') return i; + } + return -1; + } + + /** A single `new Foo(args)` constructor site (`kind: 'new'`). */ + private visitNew(node: SyntaxNode, acc: FactAccumulator): void { + const typeId = node.namedChildren.find((c) => c.type === 'type_identifier'); + const argsNode = node.namedChildren.find((c) => c.type === 'arguments'); + // `new_expression` is NOT captured for CALLS by the Dart scope-resolution, so + // this `at` finds no resolved id — anchor on the node start for consistency. + const siteIdx = acc.openCallSite('new', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); + acc.pushFrame(siteIdx); + if (typeId) acc.setSiteCallee(siteIdx, typeId.text); // the type is not a scalar binding + const resultDefs = this.resultDefTargets.get(node.id); + if (resultDefs !== undefined) acc.setSiteResultDefs(siteIdx, resultDefs); + if (argsNode) this.walkArgumentsNode(argsNode, siteIdx, acc); + acc.popFrame(); + } + + /** A cascade call `a..method(args)` — a FREE call on the method name. */ + private visitCascade(cascade: SyntaxNode, acc: FactAccumulator): void { + const cascadeSelector = cascade.namedChildren.find((c) => c.type === 'cascade_selector'); + const argPart = cascade.namedChildren.find((c) => c.type === 'argument_part'); + // A property cascade (`..field = x`, no argument_part) is not a call — walk + // its non-selector children for uses and stop. + if (!argPart) { + for (const c of cascade.namedChildren) { + if (c.type === 'cascade_selector') continue; + this.walkValue(c, acc); + } + return; + } + const nameId = cascadeSelector + ? (this.selectorName(cascadeSelector) ?? cascadeSelector) + : undefined; + const anchor: [number, number] = nameId + ? [nameId.startPosition.row + 1, nameId.startPosition.column] + : [cascade.startPosition.row + 1, cascade.startPosition.column]; + const siteIdx = acc.openCallSite('call', anchor); + acc.pushFrame(siteIdx); + if (nameId) acc.setSiteCallee(siteIdx, nameId.text); + this.walkArguments(argPart, siteIdx, acc); + acc.popFrame(); + } + + /** Walk a `selector → argument_part → arguments` for per-arg occurrences. */ + private walkArguments(argPart: SyntaxNode, siteIdx: number, acc: FactAccumulator): void { + const args = argPart.namedChildren.find((c) => c.type === 'arguments'); + if (args) this.walkArgumentsNode(args, siteIdx, acc); + } + + /** Walk an `arguments` node, tagging each positional / named arg's occurrence position. */ + private walkArgumentsNode(args: SyntaxNode, siteIdx: number, acc: FactAccumulator): void { + let pos = 0; + for (let i = 0; i < args.namedChildCount; i++) { + const arg = args.namedChild(i); + if (!arg || COMMENT_TYPES.has(arg.type)) continue; + if (arg.type === 'argument' || arg.type === 'named_argument') { + acc.setFrameArg(pos); + // The argument value may be a flattened postfix run (`escape(x)` ⇒ + // `escape` + `selector(x)`) — group via `walkChildren` so a NESTED call is + // its own parent-linked site (the via-tagged sanitizer-interposition + // substrate). A `named_argument` (`k: v`) records only the VALUE + // occurrence — the `label` name (`k`) is dropped. + const valueChildren = arg.namedChildren.filter((c) => c.type !== 'label'); + this.walkChildren(valueChildren, acc); + pos++; + } else { + // A spread `...xs` argument variant, if the grammar surfaces one. + acc.setFrameArg(pos); + acc.setSiteSpread(siteIdx, pos); + this.walkValue(arg, acc); + pos++; + } + } + } + + /** + * Register result-defs for a single-target binding whose value RUN's terminal + * call / `new` should carry `[x]`: `var x = f()` / `var x = obj.m()` / + * `var x = new Foo()` / `x = g(y)`. Keyed so the run's call selector (whose + * `.parent` is the run's shared parent — the `initialized_variable_definition` + * or `assignment_expression`) AND a single `new_expression` run-node both hit. + */ + private registerRunResultDefs(run: readonly SyntaxNode[], targets: readonly SyntaxNode[]): void { + const defs: number[] = []; + for (const target of targets) { + if (target.type !== 'identifier' || target.text === '_') continue; + defs.push(this.resolve(target)); + } + if (defs.length === 0) return; + // A postfix-chain run: its call selector keys on the run's shared parent. + const parentId = run[0]?.parent?.id; + if (parentId !== undefined) this.resultDefTargets.set(parentId, defs); + // A single `new Foo(…)` value: `visitNew` keys on the `new_expression` itself. + if (run.length === 1 && run[0].type === 'new_expression') { + this.resultDefTargets.set(run[0].id, defs); + } + } + + /** The `field`-tagged children of `node` (a Dart postfix run flattens here). */ + private fieldRun(node: SyntaxNode, field: string): SyntaxNode[] { + const run: SyntaxNode[] = []; + for (let i = 0; i < node.childCount; i++) { + const c = node.child(i); + if (!c || COMMENT_TYPES.has(c.type)) continue; + if (node.fieldNameForChild?.(i) === field) run.push(c); + } + return run; + } + /** * The bare `identifier` of an `assignable_expression` lvalue WHEN it is a * scalar target (`x = …`), or undefined when it is a member / subscript write diff --git a/gitnexus/src/core/ingestion/cfg/visitors/go-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/go-harvest.ts index 59e1a25f0..606229cb3 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/go-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/go-harvest.ts @@ -67,7 +67,7 @@ */ import type { SyntaxNode } from '../../utils/ast-helpers.js'; import type { BindingEntry, StatementFacts } from '../types.js'; -import { CallSiteFactAccumulator } from './call-site-harvest.js'; +import { CallSiteFactAccumulator, finalizeChain } from './call-site-harvest.js'; import { ScopeTreeHarvester, type Scope, type FactAccumulator } from './scope-tree-harvest.js'; /** Node types that own a nested CFG — their subtrees are opaque to harvesting. */ @@ -571,7 +571,12 @@ export class GoHarvester extends ScopeTreeHarvester { private visitCall(node: SyntaxNode, acc: FactAccumulator): void { const calleeNode = node.childForFieldName('function'); const argsNode = node.childForFieldName('arguments'); - const siteIdx = acc.openCallSite('call'); + // `node` IS the call_expression — the SAME node the scope-extractor anchors + // `@reference.call.*` (its `atRange`) on (KTD7). + const siteIdx = acc.openCallSite('call', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); acc.pushFrame(siteIdx); let calleePath: string | undefined; if (calleeNode) { @@ -640,24 +645,11 @@ export class GoHarvester extends ScopeTreeHarvester { break; } } - let rootIdx: number | undefined; - let rootSegment: string | undefined; - if (cur.type === 'identifier' && cur.text !== '_') { - rootIdx = this.resolve(cur); - acc.addUse(rootIdx); - rootSegment = cur.text; - } else { - this.walkValue(cur, acc); - } - const innermost = accesses[0]; - if (rootIdx !== undefined && innermost && !(skipFinalRead && accesses.length === 1)) { - acc.addMemberRead(rootIdx, innermost); - } - const path = - rootSegment !== undefined && accesses.every((a) => a !== '') - ? [rootSegment, ...accesses].join('.') - : undefined; - return { path, rootIdx }; + // The shared terminal: root-use record + innermost member-read + path-join. + return finalizeChain(acc, cur, accesses, skipFinalRead, (t) => t === 'identifier', { + resolve: (n) => this.resolve(n), + walkRoot: (n) => this.walkValue(n, acc), + }); } } diff --git a/gitnexus/src/core/ingestion/cfg/visitors/java-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/java-harvest.ts index c1e1c7e48..44b4568ec 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/java-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/java-harvest.ts @@ -414,7 +414,12 @@ export class JavaHarvester extends ScopeTreeHarvester { const objectNode = node.childForFieldName('object'); const nameNode = node.childForFieldName('name'); const argsNode = node.childForFieldName('arguments'); - const siteIdx = acc.openCallSite('call'); + // `node` IS the method_invocation — the SAME node the scope-extractor + // anchors `@reference.call.*` (its `atRange`) on (KTD7). + const siteIdx = acc.openCallSite('call', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); acc.pushFrame(siteIdx); let receiverPath: string | undefined; if (objectNode) { @@ -444,7 +449,12 @@ export class JavaHarvester extends ScopeTreeHarvester { private visitNew(node: SyntaxNode, acc: FactAccumulator): void { const typeNode = node.childForFieldName('type'); const argsNode = node.childForFieldName('arguments'); - const siteIdx = acc.openCallSite('new'); + // `node` IS the object_creation_expression — the SAME node the + // scope-extractor anchors `@reference.call.constructor` (its `atRange`) on. + const siteIdx = acc.openCallSite('new', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); acc.pushFrame(siteIdx); if (typeNode) { // The type name is not a scalar binding — record it only as the callee diff --git a/gitnexus/src/core/ingestion/cfg/visitors/kotlin-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/kotlin-harvest.ts index 2d27b3d89..49ae192ac 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/kotlin-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/kotlin-harvest.ts @@ -1,12 +1,42 @@ /** * Kotlin def/use harvester (#2195) — the Kotlin analogue of * {@link import('./swift-harvest.js').SwiftHarvester} and the C-family / Go / - * Rust / Python harvesters. Like the Swift / Python / Rust harvesters it harvests - * NO call-site `sites[]` (the call-site taint substrate is a later step): it emits - * only the per-function binding table ({@link BindingEntry}[]) plus - * {@link StatementFacts} (defs / uses / mayDefs) via a local - * {@link FactAccumulator} with no site machinery, so the produced facts never - * carry a `sites` key. + * Rust / Python harvesters. Like the Go / Python / Dart harvesters it harvests + * the per-function binding table ({@link BindingEntry}[]) plus + * {@link StatementFacts} (defs / uses / mayDefs) AND a taint + * {@link import('../types.js').SiteRecord} per call (callee path, receiver, + * per-arg occurrence entries, result defs, spread marker, and an `at` anchor) + * via the shared {@link CallSiteFactAccumulator} — the same site substrate the + * C-family / Go / TS / Python / Dart harvesters emit (#2227 follow-up). + * + * KOTLIN CALL SHAPE (verified by a real parse — see below). A call is a + * `call_expression` whose LAST child is a `call_suffix` (holding the + * `value_arguments` and/or a trailing `annotated_lambda`); the callee is the + * preceding expression — a bare `simple_identifier` (`foo()`) for a FREE call, + * or a `navigation_expression` (`obj.method` / `a?.b` via `navigation_suffix`) + * for a MEMBER call. A chained call `a.b.c()` nests `navigation_expression`s; + * the receiver is the chain ROOT binding. Kotlin constructor calls look like + * ordinary calls (no `new`), so every site is `kind: 'call'` (the CALLS query + * classifies a capitalized/known-type callee as `@reference.call.constructor`, + * but the harvester only needs callee + receiver + `at` right — `kind` is not + * joined). Named args (`name = value`) record the VALUE occurrence and drop the + * name (like Python / Dart). + * + * ANCHOR ALIGNMENT (plan KTD7 — load-bearing): a call site's `at` MUST be the + * SAME `[line (1-based), col (0-based)]` the Kotlin CALLS resolution keys its + * `atRange` on, because a downstream unit joins the two by EXACT position. The + * Kotlin scope query (query.ts) anchors `@reference.call.free` and + * `@reference.call.member` on the WHOLE `call_expression` node (the + * `@reference.name` simple_identifier and the `@reference.receiver` are SUB-tags, + * excluded from the anchor by `KNOWN_SUB_TAGS` + the broadest-span rule in + * `anchorCaptureFor`; `atRange: anchor.range` at scope-extractor.ts:1030). So for + * a free call `foo(x)`, a member call `obj.method(x)`, and a chained call + * `a.b.c(x)` alike, `at` is the start of the enclosing `call_expression` node — + * which, for a member/chained call, starts at the RECEIVER (`obj`/`a`), exactly + * where the CALLS anchor starts too. This is the Go/Python whole-call-node model, + * NOT the Dart callee-name model. The harvester's `visitCall` receives exactly + * the `call_expression` node and records `[node.startPosition.row + 1, + * node.startPosition.column]`. * * Runs in the parse worker next to the Kotlin CFG visitor. Output is the binding * table the {@link import('../cfg-builder.js').CfgBuilder} stamps onto the CFG, @@ -80,7 +110,7 @@ */ import type { SyntaxNode } from '../../utils/ast-helpers.js'; import type { BindingEntry, StatementFacts } from '../types.js'; -import { DefUseAccumulator as FactAccumulator } from './call-site-harvest.js'; +import { CallSiteFactAccumulator as FactAccumulator, finalizeChain } from './call-site-harvest.js'; /** Node types that own a nested CFG — their subtrees are opaque to harvesting. */ const NESTED_FUNCTION_TYPES = new Set([ @@ -99,6 +129,13 @@ export class KotlinHarvester { private readonly fnId: number; /** >0 while walking a conditionally-evaluated subexpression — defs become may-defs. */ private conditionalDepth = 0; + /** + * `call_expression` node id → binding indices its single-target result is + * assigned to (`val x = f()` / `x = g()` ⇒ `[x]`). Populated just before the + * value walk reaches the call (see {@link registerResultDefs}) and consumed by + * {@link visitCall}. Mirrors the Go / Python / Dart harvesters' `resultDefTargets`. + */ + private readonly resultDefTargets = new Map(); constructor(private readonly fnNode: SyntaxNode) { this.fnId = fnNode.id; @@ -431,6 +468,14 @@ export class KotlinHarvester { (c) => c.type === 'variable_declaration' || c.type === 'multi_variable_declaration', ); const value = this.propertyValue(node); + // Register result-defs BEFORE the value walk so the call site (reached + // during the walk) carries them — single `variable_declaration` binder + // only (`val x = f()`); a `multi_variable_declaration` destructuring + // (`val (a, b) = p`) attaches nothing. + if (value && binder?.type === 'variable_declaration') { + const id = binder.namedChildren.find((c) => c.type === 'simple_identifier'); + if (id) this.registerResultDefs(value, [id]); + } if (value) this.walkValue(value, acc); if (binder?.type === 'variable_declaration') this.defVariableDeclaration(binder, acc); else if (binder?.type === 'multi_variable_declaration') { @@ -444,9 +489,16 @@ export class KotlinHarvester { const lvalue = node.namedChildren.find((c) => c.type === 'directly_assignable_expression'); const op = this.assignmentOperator(node); const value = this.assignmentValue(node); + const scalar = lvalue ? this.unwrapAssignable(lvalue) : undefined; + // Plain `x = f(a)` attaches `resultDefs: [x]` (a compound `x += f(a)` + // does not — the prior value flows in too; a member/index lvalue is not + // a scalar def). + if (value && op === '=' && scalar?.type === 'simple_identifier') { + this.registerResultDefs(value, [scalar]); + } if (value) this.walkValue(value, acc); if (lvalue) { - const lv = this.unwrapAssignable(lvalue); + const lv = scalar ?? this.unwrapAssignable(lvalue); if (lv.type === 'simple_identifier') { this.def(lv, acc); if (op !== '=') this.use(lv, acc); // compound assign reads too @@ -471,11 +523,18 @@ export class KotlinHarvester { } return; } + case 'call_expression': + // #2227 follow-up: explicit case (previously default-descended) — same + // uses, plus a taint-site record. Kotlin has no `new` (constructor calls + // are plain `call_expression`s). Defs/uses stay byte-identical. + this.visitCall(node, acc); + return; case 'navigation_expression': { // `a.b` / `a?.b` — value read of the chain root only; the suffix name is - // not a scalar binding. - const target = node.namedChild(0); - if (target) this.walkValue(target, acc); + // not a scalar binding. Records the chain-root use (identical to the old + // descent) plus at most ONE member-read site (the innermost access), + // mirroring the Go / Python harvesters' value-position `walkChain`. + this.walkChain(node, acc, false); return; } case 'conjunction_expression': @@ -572,4 +631,180 @@ export class KotlinHarvester { } return n; } + + // ── taint-site harvest (#2227 follow-up) ───────────────────────────────── + + /** Strip `parenthesized_expression` wrappers around a value (`(f())`). */ + private unwrapValue(node: SyntaxNode): SyntaxNode { + let n = node; + let hops = 8; + while (n.type === 'parenthesized_expression' && hops-- > 0) { + const inner = n.namedChild(0); + if (!inner) break; + n = inner; + } + return n; + } + + /** + * When `value`'s root (after stripping parens) is a `call_expression`, remember + * that call site should carry `resultDefs` — the binding indices of `targets` + * (def-position identifiers). Consumed by {@link visitCall} once the value walk + * reaches the node. Single-target only (the caller restricts to a plain + * identifier binder); the blank target (`_`) binds nothing and is skipped. + */ + private registerResultDefs(value: SyntaxNode, targets: readonly SyntaxNode[]): void { + const root = this.unwrapValue(value); + if (root.type !== 'call_expression') return; + const defs: number[] = []; + for (const target of targets) { + if (target.type !== 'simple_identifier' || target.text === '_') continue; + defs.push(this.resolve(target)); + } + if (defs.length > 0) this.resultDefTargets.set(root.id, defs); + } + + /** + * The callee node of a `call_expression` — the first named child that is NOT + * the trailing `call_suffix` (a bare `simple_identifier` for a free call, or a + * `navigation_expression` for a member/chained call). + */ + private calleeOf(call: SyntaxNode): SyntaxNode | undefined { + for (let i = 0; i < call.namedChildCount; i++) { + const c = call.namedChild(i); + if (c && c.type !== 'call_suffix' && !COMMENT_TYPES.has(c.type)) return c; + } + return undefined; + } + + /** + * Explicit `call_expression` handler. Records a call site (callee path, + * receiver, per-arg occurrence entries, result defs, spread marker) while + * reproducing EXACTLY the uses the old default descent recorded (callee chain + * root + arguments). Kotlin has no `new` — every site is `kind: 'call'`. + */ + private visitCall(node: SyntaxNode, acc: FactAccumulator): void { + const calleeNode = this.calleeOf(node); + // `node` IS the `call_expression` — the SAME node the scope query anchors + // `@reference.call.free/.member` on (its `atRange`), so the resolved-id join + // lands by exact position (see file header ANCHOR ALIGNMENT). + const siteIdx = acc.openCallSite('call', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); + acc.pushFrame(siteIdx); + if (calleeNode) { + const callee = this.unwrapValue(calleeNode); + if (callee.type === 'simple_identifier') { + // A bare free call — the callee NAME is a statement-level use but NOT a + // value occurrence in any enclosing argument. + if (callee.text !== '_') acc.addUseWithoutOccurrence(this.resolve(callee)); + acc.setSiteCallee(siteIdx, callee.text); + } else if (callee.type === 'navigation_expression') { + // skipFinalRead: the final `.name` IS the callee, carried by the path. + const chain = this.walkChain(callee, acc, true); + if (chain.path !== undefined) acc.setSiteCallee(siteIdx, chain.path); + if (chain.rootIdx !== undefined) acc.setSiteReceiver(siteIdx, chain.rootIdx); + } else { + // Call-rooted chains (`f().g()`), indexing (`m[k]()`), parenthesized + // callables — the walk still records uses and nested sites; the callee + // path is not statically known. + this.walkValue(callee, acc); + } + } + const resultDefs = this.resultDefTargets.get(node.id); + if (resultDefs !== undefined) acc.setSiteResultDefs(siteIdx, resultDefs); + const suffix = node.namedChildren.find((c) => c.type === 'call_suffix'); + if (suffix) this.walkArguments(suffix, siteIdx, acc); + acc.popFrame(); + } + + /** + * Walk a `call_suffix`'s `value_arguments`, tagging each positional / named / + * spread argument's occurrence position. A trailing `annotated_lambda` is a + * nested function body — opaque (its `lambda_literal` is excluded by + * {@link NESTED_FUNCTION_TYPES}), so it is not an argument occurrence here. + */ + private walkArguments(suffix: SyntaxNode, siteIdx: number, acc: FactAccumulator): void { + const args = suffix.namedChildren.find((c) => c.type === 'value_arguments'); + if (!args) return; + let pos = 0; + for (let i = 0; i < args.namedChildCount; i++) { + const arg = args.namedChild(i); + if (!arg || arg.type !== 'value_argument') continue; + acc.setFrameArg(pos); + const value = this.argumentValue(arg); + if (value?.type === 'spread_expression') { + // `f(*xs)` — a spread. Mark the first spread position so the matcher + // degrades soundly; the inner value still walks for occurrences. + acc.setSiteSpread(siteIdx, pos); + const inner = value.namedChild(0); + if (inner) this.walkValue(inner, acc); + } else if (value) { + this.walkValue(value, acc); + } + pos++; + } + } + + /** + * The value expression of a `value_argument` — for a named argument + * (`name = value`) the leading `simple_identifier` name is dropped (it is a + * parameter name, not a use), and only the value after `=` is returned; a + * positional argument's value is its sole non-comment named child. + */ + private argumentValue(arg: SyntaxNode): SyntaxNode | undefined { + // A named arg carries an anon `=` token; the value is the named child after + // it (skipping the leading `simple_identifier` name). + let sawEq = false; + for (let i = 0; i < arg.childCount; i++) { + const c = arg.child(i); + if (!c) continue; + if (!c.isNamed && c.type === '=') { + sawEq = true; + continue; + } + if (sawEq && c.isNamed && !COMMENT_TYPES.has(c.type)) return c; + } + // Positional arg — the first non-comment named child. + for (let i = 0; i < arg.namedChildCount; i++) { + const c = arg.namedChild(i); + if (c && c.type !== 'annotation' && !COMMENT_TYPES.has(c.type)) return c; + } + return undefined; + } + + /** + * `navigation_expression` chain walk shared by value position and callee + * position. Records the chain-root identifier as a use (identical to the old + * default descent) plus at most ONE member-read site — the INNERMOST access — + * when the root is an identifier; `skipFinalRead` suppresses it when that + * access is the callee (carried by the dotted path instead). Mirrors the Go / + * Python harvesters' `walkChain`. + */ + private walkChain( + node: SyntaxNode, + acc: FactAccumulator, + skipFinalRead: boolean, + ): { path?: string; rootIdx?: number } { + const accesses: string[] = []; + let cur: SyntaxNode = this.unwrapValue(node); + for (;;) { + if (cur.type === 'navigation_expression') { + const suffix = cur.namedChildren.find((c) => c.type === 'navigation_suffix'); + const name = suffix?.namedChildren.find((c) => c.type === 'simple_identifier'); + accesses.unshift(name?.text ?? ''); + const operand = cur.namedChild(0); + if (!operand) break; + cur = this.unwrapValue(operand); + } else { + break; + } + } + // The shared terminal: root-use record + innermost member-read + path-join. + return finalizeChain(acc, cur, accesses, skipFinalRead, (t) => t === 'simple_identifier', { + resolve: (n) => this.resolve(n), + walkRoot: (n) => this.walkValue(n, acc), + }); + } } diff --git a/gitnexus/src/core/ingestion/cfg/visitors/php-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/php-harvest.ts index 8e4062457..ef8b9db4b 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/php-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/php-harvest.ts @@ -553,7 +553,12 @@ export class PhpHarvester { shape: 'function' | 'member' | 'scoped', ): void { const argsNode = node.childForFieldName('arguments'); - const siteIdx = acc.openCallSite('call'); + // `node` IS the call expression — the SAME node the scope-extractor anchors + // `@reference.call.*` (its `atRange`) on (KTD7). + const siteIdx = acc.openCallSite('call', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); acc.pushFrame(siteIdx); if (shape === 'function') { @@ -603,7 +608,12 @@ export class PhpHarvester { /** Explicit `object_creation_expression` (`new Foo($x)`) handler. */ private visitNew(node: SyntaxNode, acc: FactAccumulator): void { const argsNode = node.childForFieldName('arguments'); - const siteIdx = acc.openCallSite('new'); + // `node` IS the object_creation_expression — the SAME node the + // scope-extractor anchors `@reference.call.constructor` (its `atRange`) on. + const siteIdx = acc.openCallSite('new', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); acc.pushFrame(siteIdx); // The class name is the first `name`/`qualified_name` child (not a binding). const className = node.namedChildren.find( diff --git a/gitnexus/src/core/ingestion/cfg/visitors/python-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/python-harvest.ts index f5a4863c6..c272544b8 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/python-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/python-harvest.ts @@ -11,9 +11,14 @@ * per-statement variable definition/use facts that ride the side channel for the * reaching-defs / CDG solvers. Output is the per-function binding table * ({@link BindingEntry}[]) plus {@link StatementFacts} the visitor attaches to - * blocks as it walks. NO `sites[]` are harvested here — the call-site taint - * substrate is a later step (this unit emits only bindings + defs/uses + mayDefs - * via the local {@link FactAccumulator}, which has no site machinery at all). + * blocks as it walks. Each `call` ALSO records a taint {@link SiteRecord} (callee + * path, receiver, per-arg occurrence entries, result defs, spread marker, and an + * `at` anchor) via the shared {@link CallSiteFactAccumulator} — the same site + * substrate the C-family / Go / TS harvesters emit. Python has NO `new` + * expression (constructors are plain `call`s), so every site is `kind: 'call'`. + * The `at` anchor is the `call` node's start position, byte-aligned with the + * `@reference.call.*` CALLS-edge anchor (which also captures the whole `call` + * node), so the downstream resolved-callee-id join lands by exact position. * * Every node type and field literal below was grammar-validated against * tree-sitter-python (0.23.x) via the introspection probe before use (mandatory @@ -92,7 +97,7 @@ */ import type { SyntaxNode } from '../../utils/ast-helpers.js'; import type { BindingEntry, StatementFacts } from '../types.js'; -import { DefUseAccumulator as FactAccumulator } from './call-site-harvest.js'; +import { CallSiteFactAccumulator as FactAccumulator, finalizeChain } from './call-site-harvest.js'; /** Node types that own a nested CFG — their subtrees are opaque to harvesting. */ const NESTED_FUNCTION_TYPES = new Set(['function_definition', 'lambda']); @@ -110,6 +115,13 @@ export class PythonHarvester { private readonly fnId: number; /** >0 while walking a conditionally-evaluated subexpression — defs become may-defs. */ private conditionalDepth = 0; + /** + * `call` node id → binding indices its single-target result is assigned to + * (`x = f()` ⇒ `[x]`). Populated just before the value walk reaches the call + * (see {@link registerResultDefs}) and consumed by {@link visitCall}. Mirrors + * the Go harvester's `resultDefTargets`. + */ + private readonly resultDefTargets = new Map(); constructor(private readonly fnNode: SyntaxNode) { this.fnId = fnNode.id; @@ -542,6 +554,10 @@ export class PythonHarvester { case 'assignment': { const left = node.childForFieldName('left'); const right = node.childForFieldName('right'); + // Register result-defs BEFORE walking the value so the nested call site + // (reached during the value walk) carries them — single identifier + // target only (`x = f()`); a tuple/attribute LHS attaches nothing. + if (left?.type === 'identifier' && right) this.registerResultDefs(right, [left]); if (right) this.walkValue(right, acc); if (left) this.defTargets(left, acc); return; @@ -565,6 +581,7 @@ export class PythonHarvester { // walrus `(n := v)` — `n` is a def, `v` a use. const name = node.childForFieldName('name'); const value = node.childForFieldName('value'); + if (name?.type === 'identifier' && value) this.registerResultDefs(value, [name]); if (value) this.walkValue(value, acc); if (name?.type === 'identifier') this.def(name, acc); return; @@ -590,9 +607,10 @@ export class PythonHarvester { } case 'attribute': { // `a.b` — value read of the operand root only; the attribute name is not - // a scalar binding. - const obj = node.childForFieldName('object'); - if (obj) this.walkValue(obj, acc); + // a scalar binding. Records the chain-root use plus at most ONE + // member-read site (the innermost identifier-rooted access), mirroring + // the Go harvester's value-position `walkChain`. + this.walkChain(node, acc, false); return; } case 'subscript': { @@ -603,15 +621,12 @@ export class PythonHarvester { if (sub) this.walkValue(sub, acc); return; } - case 'call': { - // Reproduce default-descent uses: callee chain + arguments. No site - // record (taint substrate is a later step). - const fn = node.childForFieldName('function'); - const args = node.childForFieldName('arguments'); - if (fn) this.walkValue(fn, acc); - if (args) this.walkValue(args, acc); + case 'call': + // #2227 follow-up: explicit case (previously default-descended) — same + // uses, plus a taint-site record. Python has no `new` (constructor calls + // are plain `call`s). Defs/uses stay byte-identical. + this.visitCall(node, acc); return; - } case 'list_comprehension': case 'set_comprehension': case 'dictionary_comprehension': @@ -657,4 +672,127 @@ export class PythonHarvester { } if (body) this.walkValue(body, acc); } + + // ── taint-site harvest (#2227 follow-up) ───────────────────────────────── + + /** + * When `value`'s root (after stripping parens) is a `call`, remember that + * call site should carry `resultDefs` — the binding indices of `targets` + * (def-position identifiers). Consumed by {@link visitCall} once the value + * walk reaches the node. Single-target only (the caller restricts to a plain + * identifier LHS); the blank identifier (`_`) binds nothing and is skipped. + */ + private registerResultDefs(value: SyntaxNode, targets: readonly SyntaxNode[]): void { + const root = this.unwrapValue(value); + if (root.type !== 'call') return; + const defs: number[] = []; + for (const target of targets) { + if (target.type !== 'identifier' || target.text === '_') continue; + defs.push(this.resolve(target)); + } + if (defs.length > 0) this.resultDefTargets.set(root.id, defs); + } + + /** Strip `parenthesized_expression` wrappers around a value (`(f())`). */ + private unwrapValue(node: SyntaxNode): SyntaxNode { + let n = node; + let hops = 8; + while (n.type === 'parenthesized_expression' && hops-- > 0) { + const inner = n.namedChild(0); + if (!inner) break; + n = inner; + } + return n; + } + + /** + * Explicit `call` handler. Records a call site (callee path, receiver, per-arg + * occurrence entries, result defs, spread marker) while reproducing EXACTLY + * the uses the old default descent recorded (callee chain root + arguments). + * Python has no `new` — every site is `kind: 'call'`. + */ + private visitCall(node: SyntaxNode, acc: FactAccumulator): void { + const calleeNode = node.childForFieldName('function'); + const argsNode = node.childForFieldName('arguments'); + // `node` IS the `call` — the SAME node the scope-extractor anchors + // `@reference.call.free/.member` on (its `atRange`), so the resolved-id join + // lands by exact position. + const siteIdx = acc.openCallSite('call', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); + acc.pushFrame(siteIdx); + if (calleeNode) { + const callee = this.unwrapValue(calleeNode); + if (callee.type === 'identifier') { + if (callee.text !== '_') acc.addUseWithoutOccurrence(this.resolve(callee)); + acc.setSiteCallee(siteIdx, callee.text); + } else if (callee.type === 'attribute') { + // skipFinalRead: the final `.attr` IS the callee, carried by the path. + const chain = this.walkChain(callee, acc, true); + if (chain.path !== undefined) acc.setSiteCallee(siteIdx, chain.path); + if (chain.rootIdx !== undefined) acc.setSiteReceiver(siteIdx, chain.rootIdx); + } else { + // Call-rooted chains (`a().b()`), subscripts (`d[k]()`) — the walk still + // records uses and nested sites; the callee path is not statically known. + this.walkValue(callee, acc); + } + } + const resultDefs = this.resultDefTargets.get(node.id); + if (resultDefs !== undefined) acc.setSiteResultDefs(siteIdx, resultDefs); + if (argsNode) { + let pos = 0; + for (let i = 0; i < argsNode.namedChildCount; i++) { + const arg = argsNode.namedChild(i); + if (!arg || arg.type === 'comment') continue; + if (arg.type === 'list_splat' || arg.type === 'dictionary_splat') { + // `f(*args)` / `f(**kw)` — a spread. Mark the first spread position so + // the matcher degrades soundly; the inner value still walks. + acc.setFrameArg(pos); + acc.setSiteSpread(siteIdx, pos); + const inner = arg.namedChild(0); + if (inner) this.walkValue(inner, acc); + } else { + // Positional or `keyword_argument` — the `keyword_argument` case in + // `walkValue` walks only its `value`, so the key name is not a use. + acc.setFrameArg(pos); + this.walkValue(arg, acc); + } + pos++; + } + } + acc.popFrame(); + } + + /** + * `attribute` chain walk shared by value position and callee position. Records + * the chain-root identifier as a use (identical to the old default descent) + * plus at most ONE member-read site — the INNERMOST access — when the root is + * an identifier; `skipFinalRead` suppresses it when that access is the callee + * (carried by the dotted path instead). Mirrors the Go harvester's `walkChain`. + */ + private walkChain( + node: SyntaxNode, + acc: FactAccumulator, + skipFinalRead: boolean, + ): { path?: string; rootIdx?: number } { + const accesses: string[] = []; + let cur: SyntaxNode = this.unwrapValue(node); + for (;;) { + if (cur.type === 'attribute') { + const field = cur.childForFieldName('attribute'); + accesses.unshift(field?.text ?? ''); + const operand = cur.childForFieldName('object'); + if (!operand) break; + cur = this.unwrapValue(operand); + } else { + break; + } + } + // The shared terminal: root-use record + innermost member-read + path-join. + return finalizeChain(acc, cur, accesses, skipFinalRead, (t) => t === 'identifier', { + resolve: (n) => this.resolve(n), + walkRoot: (n) => this.walkValue(n, acc), + }); + } } diff --git a/gitnexus/src/core/ingestion/cfg/visitors/ruby-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/ruby-harvest.ts index 8c0faedf9..8e35b3ed4 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/ruby-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/ruby-harvest.ts @@ -3,11 +3,46 @@ * {@link import('./python-harvest.js').PythonHarvester} (the closest structural * sibling: implicit/keyword-delimited blocks, statement-modifier forms, a * begin/rescue/else/ensure exception model, and `case`/`when` + `case`/`in` - * pattern matching). Like the Python harvester, this unit emits ONLY the + * pattern matching). Like the Python / Kotlin harvesters, this unit emits the * per-function binding table ({@link BindingEntry}[]) plus {@link StatementFacts} - * (defs / uses / mayDefs) — NO call-site `sites[]` are harvested (the taint - * substrate is a later step), so it uses a local {@link FactAccumulator} with no - * site machinery at all and the emitted facts carry no `sites` key. + * (defs / uses / mayDefs) AND a taint {@link import('../types.js').SiteRecord} + * per call (callee path, receiver, per-arg occurrence entries, result defs, + * spread marker, and an `at` anchor) via the shared + * {@link CallSiteFactAccumulator} — the same site substrate the C-family / Go / + * TS / Python / Kotlin / Dart harvesters emit. + * + * RUBY CALL SHAPE (verified by a real parse — see below). EVERY call in Ruby is + * a single `call` node (fields `receiver`?/`method`/`arguments`?): a free call + * `foo(a)`, an implicit-receiver paren-less command `puts x` / `attr_accessor :x`, + * a member call `obj.method(x)`, a safe-navigation call `obj&.m()`, AND a chained + * `a.b.c` (nested `call` receivers) are all `call` nodes. There is NO separate + * `command` / `command_call` node in this vendored grammar — paren-less commands + * normalize to `call` with a `method` + `arguments` and no `receiver`. Ruby has + * NO `new` keyword (`Foo.new` is an ordinary member `call` with method `new`), so + * every site is `kind: 'call'`. A receiver-only no-args `call` (`obj.field`, + * `a.b`) is grammatically INDISTINGUISHABLE from a paren-less zero-arg member + * call, and the Ruby CALLS query tags it `@reference.call.member` too — so it is + * harvested as a `kind: 'call'` site (NOT a member-read), keeping the harvest + * byte-aligned with what the resolver assigns a callee id. Named (symbol-keyed) + * args (`f(k: v)` ⇒ a `pair` of `hash_key_symbol` + value) record the VALUE + * occurrence and drop the key (like Python / Kotlin / Dart). A `do … end` / `{ }` + * block is a nested function (opaque) — it is NOT an argument occurrence (the + * `block` field is never walked here). + * + * ANCHOR ALIGNMENT (plan KTD7 — load-bearing): a call site's `at` MUST be the + * SAME `[line (1-based), col (0-based)]` the Ruby CALLS resolution keys its + * `atRange` on, because a downstream unit joins the two by EXACT position. The + * Ruby scope query (query.ts) anchors `@reference.call.free` and + * `@reference.call.member` on the WHOLE `call` node (the `@reference.name` method + * identifier and `@reference.receiver` are SUB-tags, excluded from the anchor by + * `KNOWN_SUB_TAGS` + the broadest-span rule in `anchorCaptureFor`). So for a free + * call `foo(x)`, an implicit command `puts x`, a member call `obj.method(x)`, and + * a chained call `a.b.c` alike, `at` is the start of the `call` node — which, for + * a member/chained call, starts at the RECEIVER (`obj`/`a`), exactly where the + * CALLS anchor starts too. This is the Go/Python/Kotlin whole-call-node model, + * NOT the Dart callee-name model. The harvester records + * `[node.startPosition.row + 1, node.startPosition.column]` of the `call` node. + * (Verified byte-exact against the real Ruby query for every shape above.) * * Runs in the parse worker next to the Ruby CFG visitor. * @@ -76,7 +111,7 @@ */ import type { SyntaxNode } from '../../utils/ast-helpers.js'; import type { BindingEntry, StatementFacts } from '../types.js'; -import { DefUseAccumulator as FactAccumulator } from './call-site-harvest.js'; +import { CallSiteFactAccumulator as FactAccumulator } from './call-site-harvest.js'; /** Node types that own a nested CFG — their subtrees are opaque to harvesting. */ const NESTED_FUNCTION_TYPES = new Set([ @@ -111,6 +146,13 @@ export class RubyHarvester { private readonly fnId: number; /** >0 while walking a conditionally-evaluated subexpression — defs become may-defs. */ private conditionalDepth = 0; + /** + * `call` node id → binding indices its single-target result is assigned to + * (`x = f()` ⇒ `[x]`). Populated just before the value walk reaches the call + * (see {@link registerResultDefs}) and consumed by {@link visitCall}. Mirrors + * the Python / Kotlin / Go harvesters' `resultDefTargets`. + */ + private readonly resultDefTargets = new Map(); constructor(private readonly fnNode: SyntaxNode) { this.fnId = fnNode.id; @@ -428,6 +470,11 @@ export class RubyHarvester { case 'assignment': { const left = node.childForFieldName('left'); const right = node.childForFieldName('right'); + // Register result-defs BEFORE walking the value so the nested call site + // (reached during the value walk) carries them — single plain-identifier + // target only (`x = f()`); a multi-target / `@ivar` / index LHS attaches + // nothing (the per-target mapping is ambiguous). + if (left?.type === 'identifier' && right) this.registerResultDefs(right, [left]); if (right) this.walkValue(right, acc); if (left) this.defTargets(left, acc); return; @@ -462,16 +509,14 @@ export class RubyHarvester { } return; } - case 'call': { - // `recv.meth(args)` / `meth(args)` — the receiver root + arguments are - // uses (the method name is not a scalar binding). A nested block child is - // its OWN function CFG (opaque here). - const receiver = node.childForFieldName('receiver'); - const args = node.childForFieldName('arguments'); - if (receiver) this.walkValue(receiver, acc); - if (args) this.walkValue(args, acc); + case 'call': + // `recv.meth(args)` / `meth(args)` / `puts x` / `obj.field` — every Ruby + // call shape. Records a taint site (callee path, receiver, per-arg + // occurrences, result defs, spread) plus the SAME uses the old default + // descent recorded (receiver chain root + arguments). A nested block + // child is its OWN function CFG (opaque — the `block` field is not walked). + this.visitCall(node, acc); return; - } default: for (let i = 0; i < node.namedChildCount; i++) { const c = node.namedChild(i); @@ -479,4 +524,155 @@ export class RubyHarvester { } } } + + // ── taint-site harvest ──────────────────────────────────────────────────── + + /** + * When `value` is a `call`, remember that call site should carry `resultDefs` + * — the binding indices of `targets` (def-position identifiers). Consumed by + * {@link visitCall} once the value walk reaches the node. Single plain-identifier + * target only (the caller restricts to it); the blank target (`_`) binds nothing + * and is skipped. + */ + private registerResultDefs(value: SyntaxNode, targets: readonly SyntaxNode[]): void { + if (value.type !== 'call') return; + const defs: number[] = []; + for (const target of targets) { + if (target.type !== 'identifier' || target.text === '_') continue; + defs.push(this.resolve(target)); + } + if (defs.length > 0) this.resultDefTargets.set(value.id, defs); + } + + /** + * Explicit `call` handler. Records a call site (callee path, receiver, per-arg + * occurrence entries, result defs, spread marker) while reproducing EXACTLY the + * uses the old default descent recorded (receiver chain root + arguments). Ruby + * has no `new` — every site is `kind: 'call'`. The `at` anchor is the `call` + * node's start, byte-aligned with the `@reference.call.free/.member` CALLS + * anchor (which captures the whole `call` node), so the resolved-id join lands + * by exact position (see file header ANCHOR ALIGNMENT). + */ + private visitCall(node: SyntaxNode, acc: FactAccumulator): void { + const receiver = node.childForFieldName('receiver'); + const method = node.childForFieldName('method'); + const args = node.childForFieldName('arguments'); + const siteIdx = acc.openCallSite('call', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); + acc.pushFrame(siteIdx); + if (receiver) { + // Member / chained call (`obj.method`, `a.b.c`, `obj&.m`). Walk the receiver + // chain to its root binding (the receiver use + any mid-chain member reads) + // and build the dotted callee path `root.…​.method`. + const chain = this.walkReceiverChain(receiver, acc); + if (chain.rootIdx !== undefined) acc.setSiteReceiver(siteIdx, chain.rootIdx); + const path = this.calleePath(chain.path, method); + if (path !== undefined) acc.setSiteCallee(siteIdx, path); + } else if (method) { + // Free / implicit-receiver call (`foo(a)`, `puts x`, `attr_accessor :x`). + // The method NAME is a statement-level use (a known local resolves to its + // binding, an unknown method to a `module` synthetic) but NOT a value + // occurrence in any enclosing argument. + if (method.text !== '_') acc.addUseWithoutOccurrence(this.resolve(method)); + acc.setSiteCallee(siteIdx, method.text); + } + const resultDefs = this.resultDefTargets.get(node.id); + if (resultDefs !== undefined) acc.setSiteResultDefs(siteIdx, resultDefs); + if (args) this.walkArguments(args, siteIdx, acc); + acc.popFrame(); + } + + /** + * Walk a member-call receiver, returning the chain ROOT binding (recorded as a + * use) and the dotted prefix path (`a.b` in `a.b.c()`), plus a member-read site + * for each NON-final access in the chain. A receiver that is itself a `call` + * (Ruby nests `a.b.c` as `call(call(a,b),c)`) recurses; a bare identifier root + * is the receiver binding; a `self` / non-identifier root has no static path. + */ + private walkReceiverChain( + receiver: SyntaxNode, + acc: FactAccumulator, + ): { rootIdx?: number; path?: string } { + // Collect the access names along the receiver chain (outermost-last), and the + // chain root node, by unwinding nested `call` receivers. + const accesses: string[] = []; + let cur = receiver; + for (;;) { + if (cur.type === 'call' && cur.childForFieldName('arguments') === null) { + // A no-args member `call` in receiver position is a member ACCESS + // (`a.b` in `a.b.c`) — its method name extends the path; recurse on its + // own receiver. A receiver `call` WITH arguments is an opaque call result + // (`foo(a).bar` — handled by the `else` below as a nested call site). + const m = cur.childForFieldName('method'); + const inner = cur.childForFieldName('receiver'); + accesses.unshift(m?.text ?? ''); + if (!inner) break; + cur = inner; + continue; + } + break; + } + let rootIdx: number | undefined; + let rootSegment: string | undefined; + if (cur.type === 'identifier' && cur.text !== '_') { + rootIdx = this.resolve(cur); + acc.addUse(rootIdx); + rootSegment = cur.text; + } else { + // `self` / `@ivar` / a call-rooted receiver (`foo(a).bar`) / literal — walk + // for uses + nested sites; no static root segment. + this.walkValue(cur, acc); + } + // The INNERMOST access (`a.b` in `a.b.c()`) is a value-position member read; + // the trailing access (the receiver's own method) is part of the callee path, + // not a separate read. + if (rootIdx !== undefined && accesses.length >= 1) { + acc.addMemberRead(rootIdx, accesses[0]); + } + const path = + rootSegment !== undefined && accesses.every((a) => a !== '') + ? [rootSegment, ...accesses].join('.') + : undefined; + return { rootIdx, path }; + } + + /** Dotted callee path `prefix.method` (or undefined when the prefix is opaque). */ + private calleePath(prefix: string | undefined, method: SyntaxNode | null): string | undefined { + const m = method?.text; + if (m === undefined || m.length === 0) return undefined; + if (prefix === undefined) return undefined; + return `${prefix}.${m}`; + } + + /** + * Walk an `argument_list`, tagging each positional / keyword / splat argument's + * occurrence position. A `pair` (`k: v`) records only the VALUE occurrence (the + * `hash_key_symbol` key is not a use). A `splat_argument` (`*xs`) / + * `hash_splat_argument` (`**kw`) marks the first spread position so the matcher + * degrades soundly; its inner value still walks. A `block_argument` (`&blk`) + * passes a block — its inner value is a use occurrence. + */ + private walkArguments(args: SyntaxNode, siteIdx: number, acc: FactAccumulator): void { + let pos = 0; + for (let i = 0; i < args.namedChildCount; i++) { + const arg = args.namedChild(i); + if (!arg || arg.type === 'comment') continue; + acc.setFrameArg(pos); + if (arg.type === 'splat_argument' || arg.type === 'hash_splat_argument') { + acc.setSiteSpread(siteIdx, pos); + const inner = arg.namedChild(0); + if (inner) this.walkValue(inner, acc); + } else if (arg.type === 'pair') { + // `k: v` — only the value is an occurrence; the symbol key is not a use. + const value = arg.childForFieldName('value') ?? arg.namedChild(arg.namedChildCount - 1); + if (value) this.walkValue(value, acc); + } else { + // Positional arg, `block_argument` (`&blk`), or a nested expression. + this.walkValue(arg, acc); + } + pos++; + } + } } diff --git a/gitnexus/src/core/ingestion/cfg/visitors/rust-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/rust-harvest.ts index baccb7411..fc810bfa5 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/rust-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/rust-harvest.ts @@ -1,11 +1,82 @@ /** * Rust def/use harvester (#2195 U7) — the Rust analogue of * {@link import('./typescript-harvest.js').TsHarvester} and the C-family / - * Go / Python harvesters. Like the Python harvester it harvests NO call-site - * `sites[]` (the call-site taint substrate is a later step): it emits only the - * per-function binding table ({@link BindingEntry}[]) plus {@link StatementFacts} - * (defs / uses / mayDefs) via a local {@link FactAccumulator} with no site - * machinery, so the produced facts never carry a `sites` key. + * Go / Python / Swift / Kotlin / Dart harvesters. Like the Swift / Kotlin / Go / + * Python / Dart harvesters it harvests the per-function binding table + * ({@link BindingEntry}[]) plus {@link StatementFacts} (defs / uses / mayDefs) + * AND a taint {@link import('../types.js').SiteRecord} per call (callee path, + * receiver, per-arg occurrence entries, result defs, and an `at` anchor) via the + * shared {@link CallSiteFactAccumulator} — the same site substrate the + * C-family / Go / TS / Kotlin / Python / Dart / Swift harvesters emit, so Rust + * BasicBlocks get `callees` + `calleeIds`. + * + * RUST CALL SHAPE (verified by a real parse — see the probe table below). Rust + * has ONE call node, `call_expression { function, arguments }`, whose `function` + * field takes three shapes: + * 1. a bare `identifier` (`foo(x)`) — a FREE call; callee path = the name. + * 2. a `field_expression { value, field }` (`a.method(x)`) — a METHOD call + * (the `.` access). The dotted path is `a.method` (leaf `method`); the + * receiver is the chain ROOT binding (`a`). Chained `a.b.c()` nests + * `field_expression`s (path `a.b.c`, root `a`, mid-chain read `a.b`). + * 3. a `scoped_identifier { path, name }` (`Foo::bar(x)` / `a::b::c(x)`) — an + * associated-fn / path call. The path is joined with `.` (NOT `::`) so the + * LEAF after the last separator is the tail (`Foo::bar` ⇒ `Foo.bar` ⇒ leaf + * `bar`), exactly matching the Rust CALLS query, which tags this + * `@reference.call.free` with `@reference.name` = the tail `name: + * (identifier)` (`bar`), and matching how {@link + * import('../emit.js').calleesOfBlock} extracts the leaf via + * `callee.slice(callee.lastIndexOf('.') + 1)`. The receiver is set only when + * the path ROOT is a bound local (`a::b::c` with `a` a local ⇒ receiver `a`); + * a type/module root (`Foo`, `crate`) is not a value binding, so no receiver. + * 4. a `generic_function { function, type_arguments }` (`foo::(x)`) — the + * turbofish form; `visitCall` unwraps the `function` field and recurses, so + * `foo::(x)` records the same site as `foo(x)`. + * A `try_expression` (`foo()?`) wraps a `call_expression` — the inner call walks + * normally. + * + * STRUCT LITERALS ARE HARVESTED AS `kind: 'new'` (U4). A struct-literal + * expression `Point { x: 1 }` is a `struct_expression { name, body }`, NOT a + * `call_expression`. The Rust CALLS query tags it `@reference.call.constructor` + * (resolving to a constructor id), so `visitStruct` opens a `kind: 'new'` site + * whose callee path is the struct TYPE name (`mymod::Point` ⇒ dotted `mymod.Point` + * ⇒ leaf `Point`, the SAME tail the `@reference.name` capture resolves) and whose + * `at` is the `struct_expression` start (== the broadest-span + * `@reference.call.constructor` anchor — verified byte-equal for plain / scoped / + * turbofish / scoped+turbofish forms). The `name` field shapes are + * `type_identifier` (`Point`), `scoped_type_identifier` (`mymod::Point`), and + * `generic_type_with_turbofish` (`Foo::` / `mymod::Bar::`); all start at the + * same column as the `struct_expression`, so the anchor aligns. The struct's + * type/module head is no value binding ⇒ no receiver. Field-init VALUES + * (`field_initializer` `value`, shorthand `y`, base `..rest`) walk for + * uses/occurrences; field NAMES are not uses. + * + * MACROS ARE NOT HARVESTED. `println!(...)` / `vec!(...)` are `macro_invocation` + * nodes (a `macro` ident + a `token_tree`), NOT `call_expression`s. The Rust + * CALLS query tags them `@reference.macro` (a DISJOINT namespace resolved via the + * MacroRegistry to Macro defs, never a fn of the same name) — NOT + * `@reference.call.*`. So no resolved callee-id is keyed at a macro's position, + * and opening a call site there would put a leaf (`println`) into `callees` that + * the resolution side never produces — a spurious, unjoinable callee. We + * therefore record NO site for a macro (its argument identifiers still walk for + * uses via the default token-tree descent), keeping `callees` aligned with the + * CALLS resolution. + * + * ANCHOR ALIGNMENT (plan KTD7 — load-bearing): a call site's `at` MUST be the + * SAME `[line (1-based), col (0-based)]` the Rust CALLS resolution keys its + * `atRange` on, because a downstream unit joins the two by EXACT position. The + * Rust scope query (captures.ts) anchors `@reference.call.free` (free + scoped), + * `@reference.call.member`, and `@reference.call.constructor` on the WHOLE + * `call_expression` node (the `@reference.name` identifier / `field_identifier` + * and the `@reference.receiver` are SUB-tags in `KNOWN_SUB_TAGS`, excluded by the + * broadest-span rule in `anchorCaptureFor`; `atRange: anchor.range` at + * scope-extractor.ts:1030). So for a free call `foo(x)`, a method call + * `a.method(x)`, a path call `Foo::bar(x)`, and a chained call `a.b.c(x)` alike, + * `at` is the start of the enclosing `call_expression` node — which, for a + * method/chained call, starts at the RECEIVER (`a`), and for a path call at the + * head segment (`Foo`), exactly where the CALLS anchor starts too. This is the + * Swift / Go / Python / Kotlin whole-call-node model, NOT the Dart callee-name + * model. `visitCall` receives exactly the `call_expression` node and records + * `[node.startPosition.row + 1, node.startPosition.column]`. * * Runs in the parse worker next to the Rust CFG visitor. Output is the binding * table the {@link import('../cfg-builder.js').CfgBuilder} stamps onto the CFG, @@ -80,7 +151,7 @@ */ import type { SyntaxNode } from '../../utils/ast-helpers.js'; import type { BindingEntry, StatementFacts } from '../types.js'; -import { DefUseAccumulator as FactAccumulator } from './call-site-harvest.js'; +import { CallSiteFactAccumulator as FactAccumulator, finalizeChain } from './call-site-harvest.js'; /** Node types that own a nested CFG — their subtrees are opaque to harvesting. */ const NESTED_FUNCTION_TYPES = new Set(['function_item', 'closure_expression']); @@ -101,6 +172,13 @@ export class RustHarvester { private readonly fnId: number; /** >0 while walking a conditionally-evaluated subexpression — defs become may-defs. */ private conditionalDepth = 0; + /** + * `call_expression` node id → binding indices its single-target result is + * assigned to (`let x = f()` / `x = g()` ⇒ `[x]`). Populated just before the + * value walk reaches the call (see {@link registerResultDefs}) and consumed by + * {@link visitCall}. Mirrors the Swift / Kotlin / Go / Python harvesters' map. + */ + private readonly resultDefTargets = new Map(); constructor(private readonly fnNode: SyntaxNode) { this.fnId = fnNode.id; @@ -524,6 +602,10 @@ export class RustHarvester { const value = node.childForFieldName('value'); const pat = node.childForFieldName('pattern'); const alt = node.childForFieldName('alternative'); // `let … else { … }` + // Register result-defs BEFORE the value walk so the call the walk reaches + // carries them — single plain-identifier pattern only (`let x = f()`); a + // destructuring `let (a, b) = …` attaches nothing (ambiguous mapping). + if (value && pat && pat.type === 'identifier') this.registerResultDefs(value, [pat]); if (value) this.walkValue(value, acc); if (alt) this.walkValue(alt, acc); if (pat) this.defPattern(pat, acc); @@ -539,6 +621,9 @@ export class RustHarvester { case 'assignment_expression': { const left = node.childForFieldName('left'); const right = node.childForFieldName('right'); + // A plain `x = f(a)` attaches `resultDefs: [x]`; a field/index lvalue does + // not (no scalar target). + if (right && left && left.type === 'identifier') this.registerResultDefs(right, [left]); if (right) this.walkValue(right, acc); if (left) { if (left.type === 'identifier') { @@ -575,11 +660,27 @@ export class RustHarvester { } return; } + case 'call_expression': + // A Rust call (`foo(a)`, `a.method(x)`, `Foo::bar(x)`, `foo::(x)`). + // Records a taint site (callee path, receiver, per-arg occurrences, result + // defs) while reproducing the uses the old default descent recorded. Rust + // has no `new` — every site is `kind: 'call'`. + this.visitCall(node, acc); + return; + case 'struct_expression': + // A Rust struct literal (`Point { x: 1 }`, `mymod::Point { .. }`, + // `Foo:: { .. }`). The Rust CALLS query tags it + // `@reference.call.constructor`, so it resolves to a constructor id — we + // record a `kind: 'new'` site (callee = the struct type path) so that id + // joins into `calleeIds`. Field-init VALUES walk for uses/occurrences. + this.visitStruct(node, acc); + return; case 'field_expression': { - // `a.b` — value read of the chain root only; the field name is not a - // scalar binding. - const value = node.childForFieldName('value'); - if (value) this.walkValue(value, acc); + // `a.b` value read — the chain-root identifier is a use plus at most one + // member-read site (the innermost access); the field name is not a scalar + // binding. Mirrors the Swift / Kotlin / Go value-position member-read + // semantics. + this.walkChain(node, acc, false); return; } default: @@ -589,4 +690,290 @@ export class RustHarvester { } } } + + // ── taint-site harvest ─────────────────────────────────────────────────── + + /** + * When `value`'s root (after unwrapping a `try_expression`) is a + * `call_expression`, remember that call site should carry `resultDefs` — the + * binding indices of `targets` (def-position identifiers). Consumed by + * {@link visitCall} once the value walk reaches the node. Single-target only; + * the blank target (`_`) binds nothing. + */ + private registerResultDefs(value: SyntaxNode, targets: readonly SyntaxNode[]): void { + const root = this.unwrapValue(value); + if (root.type !== 'call_expression') return; + const defs: number[] = []; + for (const target of targets) { + if (target.type !== 'identifier' || target.text === '_') continue; + defs.push(this.resolve(target)); + } + if (defs.length > 0) this.resultDefTargets.set(root.id, defs); + } + + /** Strip a `try_expression` (`expr?`) / `await_expression` wrapper around a value. */ + private unwrapValue(node: SyntaxNode): SyntaxNode { + let n = node; + let hops = 4; + while ((n.type === 'try_expression' || n.type === 'await_expression') && hops-- > 0) { + const inner = n.namedChild(0); + if (!inner) break; + n = inner; + } + return n; + } + + /** + * Open + populate a call site for a Rust `call_expression`. `node` IS the + * `call_expression` — the SAME node the scope query anchors `@reference.call.*` + * on (its `atRange`), so the resolved-id join lands by exact position (see file + * header ANCHOR ALIGNMENT). A `call_expression` is always `kind: 'call'`; struct + * literals (`kind: 'new'`) are harvested separately by {@link visitStruct}. + */ + private visitCall(node: SyntaxNode, acc: FactAccumulator): void { + const calleeNode = node.childForFieldName('function'); + const argsNode = node.childForFieldName('arguments'); + const siteIdx = acc.openCallSite('call', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); + acc.pushFrame(siteIdx); + if (calleeNode) this.harvestCallee(calleeNode, siteIdx, acc); + const resultDefs = this.resultDefTargets.get(node.id); + if (resultDefs !== undefined) acc.setSiteResultDefs(siteIdx, resultDefs); + if (argsNode) { + let pos = 0; + for (let i = 0; i < argsNode.namedChildCount; i++) { + const arg = argsNode.namedChild(i); + if (!arg || arg.type === 'line_comment' || arg.type === 'block_comment') continue; + acc.setFrameArg(pos); + this.walkValue(arg, acc); + pos++; + } + } + acc.popFrame(); + } + + /** + * Open + populate a `kind: 'new'` site for a Rust `struct_expression` + * (`Point { x: 1 }`, `mymod::Point { .. }`, `Foo:: { .. }`). `node` IS the + * `struct_expression` — the SAME node the Rust scope query anchors + * `@reference.call.constructor` on (its `atRange`), so the resolved + * constructor-id join lands by exact position. The `name` field of a + * `struct_expression` is a `type_identifier` (`Point`), a + * `scoped_type_identifier` (`mymod::Point`), or a `generic_type_with_turbofish` + * (`Foo::` / `mymod::Bar::`); all three start at the SAME column as the + * enclosing `struct_expression` (verified by a real parse), so the broadest-span + * `@reference.call.constructor` anchor == the `struct_expression` start. + * + * The callee path joins the `::`-segments of the type with `.` (NOT `::`) so the + * LEAF after the last separator is the tail (`mymod::Point` ⇒ `mymod.Point` ⇒ + * leaf `Point`), exactly the tail the CALLS query's `@reference.name` capture + * resolves and the tail {@link import('../emit.js').calleesOfBlock} extracts via + * `lastIndexOf('.')`. A type/module path head is never a value binding, so no + * receiver (mirrors {@link harvestScopedCallee}). The field-init VALUES + * (`field_initializer` `value`, shorthand `y`, base `..rest`) walk for + * uses/occurrences; field NAMES are not uses. + */ + private visitStruct(node: SyntaxNode, acc: FactAccumulator): void { + const nameNode = node.childForFieldName('name'); + const bodyNode = node.childForFieldName('body'); + const siteIdx = acc.openCallSite('new', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); + acc.pushFrame(siteIdx); + const path = nameNode ? this.structTypePath(nameNode) : undefined; + if (path !== undefined) acc.setSiteCallee(siteIdx, path); + if (bodyNode) { + let pos = 0; + for (let i = 0; i < bodyNode.namedChildCount; i++) { + const field = bodyNode.namedChild(i); + if (!field) continue; + if (field.type === 'field_initializer') { + // `x: VALUE` — the field NAME is not a use; only VALUE is walked. + const value = field.childForFieldName('value'); + if (value) { + acc.setFrameArg(pos); + this.walkValue(value, acc); + pos++; + } + } else if (field.type === 'shorthand_field_initializer') { + // `y` shorthand — the identifier IS a value use of the local `y`. + const id = field.namedChild(0); + if (id) { + acc.setFrameArg(pos); + this.walkValue(id, acc); + pos++; + } + } else if (field.type === 'base_field_initializer') { + // `..rest` functional-update base — `rest` is a value use. + const baseExpr = field.namedChild(0); + if (baseExpr) { + acc.setFrameArg(pos); + this.walkValue(baseExpr, acc); + pos++; + } + } + } + } + acc.popFrame(); + } + + /** + * Build the dotted type path of a `struct_expression`'s `name` field. The name + * is a `type_identifier` (`Point`), a `scoped_type_identifier` + * (`mymod::Point` — `path` + tail `name` type_identifier), or a + * `generic_type_with_turbofish` (`Foo::` / `mymod::Bar::` — its `type` + * field is a `type_identifier` or a `scoped_identifier`; the turbofish + * `type_arguments` are dropped). Segments join with `.` so the leaf is the type + * tail (matching the CALLS `@reference.name` tail capture). Returns `undefined` + * when no segments could be read (defensive — keeps a mis-anchored site from + * carrying a bogus callee). + */ + private structTypePath(nameNode: SyntaxNode): string | undefined { + const segments: string[] = []; + const collect = (n: SyntaxNode): void => { + const t = n.type; + if (t === 'type_identifier' || t === 'identifier') { + segments.push(n.text); + return; + } + if (t === 'generic_type_with_turbofish') { + // `Foo::` / `mymod::Bar::` — descend the `type` field; the + // `type_arguments` are not part of the resolved type path. + const typeNode = n.childForFieldName('type'); + if (typeNode) collect(typeNode); + return; + } + if (t === 'scoped_type_identifier' || t === 'scoped_identifier') { + // `mymod::Point` / `mymod::Bar` — head `path` then the tail `name`. + const pathNode = n.childForFieldName('path'); + if (pathNode) collect(pathNode); + const tail = n.childForFieldName('name'); + if (tail) collect(tail); + return; + } + }; + collect(nameNode); + return segments.length > 0 && segments.every((s) => s !== '') ? segments.join('.') : undefined; + } + + /** + * Record the callee path + receiver for a `call_expression`'s `function` node. + * Free `identifier` (`foo`), method `field_expression` (`a.method`, receiver + * root `a`), path `scoped_identifier` (`Foo::bar` ⇒ dotted `Foo.bar`, leaf + * `bar`), and the turbofish `generic_function` (`foo::` — unwrap the + * `function` field and recurse). Anything else (a call-rooted chain `f()()`, + * a parenthesized callable) walks for uses with no static callee path. + */ + private harvestCallee(calleeNode: SyntaxNode, siteIdx: number, acc: FactAccumulator): void { + const callee = this.unwrapValue(calleeNode); + if (callee.type === 'identifier') { + // A bare free call — the callee NAME is a statement-level use but NOT a + // value occurrence in any enclosing argument. + if (callee.text !== '_') acc.addUseWithoutOccurrence(this.resolve(callee)); + acc.setSiteCallee(siteIdx, callee.text); + return; + } + if (callee.type === 'field_expression') { + // skipFinalRead: the final `.field` IS the callee, carried by the path. + const chain = this.walkChain(callee, acc, true); + if (chain.path !== undefined) acc.setSiteCallee(siteIdx, chain.path); + if (chain.rootIdx !== undefined) acc.setSiteReceiver(siteIdx, chain.rootIdx); + return; + } + if (callee.type === 'scoped_identifier') { + const scoped = this.harvestScopedCallee(callee, acc); + if (scoped.path !== undefined) acc.setSiteCallee(siteIdx, scoped.path); + if (scoped.rootIdx !== undefined) acc.setSiteReceiver(siteIdx, scoped.rootIdx); + return; + } + if (callee.type === 'generic_function') { + // `foo::(x)` — the turbofish wraps the real callee in `function`. + const inner = callee.childForFieldName('function'); + if (inner) this.harvestCallee(inner, siteIdx, acc); + else this.walkValue(callee, acc); + return; + } + // Call-rooted chains (`f()()`), parenthesized callables — walk for uses; no + // static callee path. + this.walkValue(callee, acc); + } + + /** + * Walk a `scoped_identifier` (`Foo::bar`, `a::b::c`) callee. The `::`-segments + * are joined with `.` (NOT `::`) so the LEAF after the last separator is the + * tail (`Foo::bar` ⇒ `Foo.bar` ⇒ leaf `bar`), matching the Rust CALLS query's + * `@reference.name` tail capture and {@link + * import('../emit.js').calleesOfBlock}'s `lastIndexOf('.')` leaf rule. The + * receiver is set only when the head segment is a bound LOCAL (`a::b::c` with + * `a` a local); a type / module head (`Foo`, `crate`) is no value binding. + */ + private harvestScopedCallee( + node: SyntaxNode, + acc: FactAccumulator, + ): { path?: string; rootIdx?: number } { + const segments: string[] = []; + let cur: SyntaxNode = node; + for (;;) { + if (cur.type === 'scoped_identifier') { + const name = cur.childForFieldName('name'); + segments.unshift(name?.text ?? ''); + const path = cur.childForFieldName('path'); + if (!path) break; + cur = path; + } else { + // The head segment — a bare `identifier` (`a` / `Foo` / `crate`) or a + // `crate`/`self`/`super`/`metavariable` keyword node. + segments.unshift(cur.text); + break; + } + } + let rootIdx: number | undefined; + // Only a head segment that is a bound LOCAL is a receiver (taint substrate); + // a type / module path head launders no value. + if (cur.type === 'identifier' && cur.text !== '_' && this.table.has(cur.text)) { + rootIdx = this.resolve(cur); + acc.addUse(rootIdx); + } + const path = segments.every((s) => s !== '') ? segments.join('.') : undefined; + return { path, rootIdx }; + } + + /** + * `field_expression` chain walk shared by value position and callee position. + * Records the chain-root identifier as a use plus at most ONE member-read site + * — the INNERMOST access — when the root is an identifier; `skipFinalRead` + * suppresses it when that access is the callee (carried by the dotted path + * instead). Mirrors the Swift / Kotlin / Go / Python harvesters' walkChain. A + * non-identifier root (`self`/literal/call) launders no static path/receiver + * but its uses + nested sites are still walked. + */ + private walkChain( + node: SyntaxNode, + acc: FactAccumulator, + skipFinalRead: boolean, + ): { path?: string; rootIdx?: number } { + const accesses: string[] = []; + let cur: SyntaxNode = node; + for (;;) { + if (cur.type === 'field_expression') { + const field = cur.childForFieldName('field'); + accesses.unshift(field?.text ?? ''); + const value = cur.childForFieldName('value'); + if (!value) break; + cur = value; + } else { + break; + } + } + // The shared terminal: root-use record + innermost member-read + path-join. + // The non-identifier root (`self.x.f()`, `foo().bar`, a tuple index) walks for + // uses + nested sites. + return finalizeChain(acc, cur, accesses, skipFinalRead, (t) => t === 'identifier', { + resolve: (n) => this.resolve(n), + walkRoot: (n) => this.walkValue(n, acc), + }); + } } diff --git a/gitnexus/src/core/ingestion/cfg/visitors/swift-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/swift-harvest.ts index e9bba8bee..8f3b615dd 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/swift-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/swift-harvest.ts @@ -1,12 +1,49 @@ /** * Swift def/use harvester (#2195) — the Swift analogue of * {@link import('./typescript-harvest.js').TsHarvester} and the C-family / Go / - * Rust / Python harvesters. Like the Python / Rust harvesters it harvests NO - * call-site `sites[]` (the call-site taint substrate is a later step): it emits - * only the per-function binding table ({@link BindingEntry}[]) plus - * {@link StatementFacts} (defs / uses / mayDefs) via a local - * {@link FactAccumulator} with no site machinery, so the produced facts never - * carry a `sites` key. + * Rust / Kotlin / Python harvesters. Like the Kotlin / Go / Python / Dart + * harvesters it harvests the per-function binding table ({@link BindingEntry}[]) + * plus {@link StatementFacts} (defs / uses / mayDefs) AND a taint + * {@link import('../types.js').SiteRecord} per call (callee path, receiver, + * per-arg occurrence entries, result defs, and an `at` anchor) via the shared + * {@link CallSiteFactAccumulator} — the same site substrate the C-family / Go / + * TS / Kotlin / Python / Dart harvesters emit. + * + * SWIFT CALL SHAPE (verified by a real parse — see below; structurally identical + * to Kotlin). A call is a `call_expression` whose LAST child is a `call_suffix` + * (holding the `value_arguments` and/or a trailing closure `lambda_literal`); the + * callee is the preceding expression — a bare `simple_identifier` (`foo(...)`) + * for a FREE call, or a `navigation_expression` (`obj.method` / `a?.b`, fields + * `target`/`suffix`→`navigation_suffix`→`suffix`:`simple_identifier`) for a + * MEMBER call. A chained call `a.b.c()` nests `navigation_expression`s; the + * receiver is the chain ROOT binding (`self`/literal roots launder no taint — + * no receiver). Swift has no `new` — an init call `Foo(...)` is an ordinary + * `call_expression` with a `simple_identifier` callee, so every site is + * `kind: 'call'` (the CALLS query re-tags an UpperCamelCase callee + * `@reference.call.constructor`, but the harvester only needs callee + receiver + * + `at` right — `kind` is not joined). A `value_argument` carries its value in + * the `value` field; a labeled arg's `value_argument_label` (`name:`) is dropped, + * so only the value occurrence is recorded (an `&inout` value walks its target + * for the use). Trailing closures (`xs.map { … }`) are a nested `lambda_literal` + * — opaque (in {@link NESTED_FUNCTION_TYPES}), NOT an argument occurrence. + * + * ANCHOR ALIGNMENT (plan KTD7 — load-bearing): a call site's `at` MUST be the + * SAME `[line (1-based), col (0-based)]` the Swift CALLS resolution keys its + * `atRange` on, because a downstream unit joins the two by EXACT position. The + * Swift scope query (query.ts) anchors `@reference.call.free`, + * `@reference.call.member`, and `@reference.call.constructor` on the WHOLE + * `call_expression` node (the `@reference.name` simple_identifier and the + * `@reference.receiver` are SUB-tags, excluded from the anchor by + * `KNOWN_SUB_TAGS` + the broadest-span rule in `anchorCaptureFor`; the + * constructor re-tag at `captures.ts` reuses the same call_expression node, and + * `atRange: anchor.range` at scope-extractor.ts:1030). So for a free call + * `foo(x)`, a member call `obj.method(x)`, a chained call `a.b.c(x)`, and an init + * call `Foo(x)` alike, `at` is the start of the enclosing `call_expression` node + * — which, for a member/chained call, starts at the RECEIVER (`obj`/`a`), exactly + * where the CALLS anchor starts too. This is the Kotlin/Go/Python whole-call-node + * model, NOT the Dart callee-name model. `visitCall` receives exactly the + * `call_expression` node and records `[node.startPosition.row + 1, + * node.startPosition.column]`. * * Runs in the parse worker next to the Swift CFG visitor. Output is the binding * table the {@link import('../cfg-builder.js').CfgBuilder} stamps onto the CFG, @@ -78,7 +115,7 @@ */ import type { SyntaxNode } from '../../utils/ast-helpers.js'; import type { BindingEntry, StatementFacts } from '../types.js'; -import { DefUseAccumulator as FactAccumulator } from './call-site-harvest.js'; +import { CallSiteFactAccumulator as FactAccumulator, finalizeChain } from './call-site-harvest.js'; /** Node types that own a nested CFG — their subtrees are opaque to harvesting. */ const NESTED_FUNCTION_TYPES = new Set([ @@ -96,6 +133,13 @@ export class SwiftHarvester { private readonly fnId: number; /** >0 while walking a conditionally-evaluated subexpression — defs become may-defs. */ private conditionalDepth = 0; + /** + * `call_expression` node id → binding indices its single-target result is + * assigned to (`let x = f()` / `x = g()` ⇒ `[x]`). Populated just before the + * value walk reaches the call (see {@link registerResultDefs}) and consumed by + * {@link visitCall}. Mirrors the Kotlin / Go / Python harvesters' map. + */ + private readonly resultDefTargets = new Map(); constructor(private readonly fnNode: SyntaxNode) { this.fnId = fnNode.id; @@ -465,14 +509,20 @@ export class SwiftHarvester { case 'property_declaration': { // Walk each `value` for uses, then def each `name` pattern's leaves. const names: SyntaxNode[] = []; + const values: SyntaxNode[] = []; for (let i = 0; i < node.childCount; i++) { const field = node.fieldNameForChild(i); const child = node.child(i); if (!child) continue; - if (field === 'value') this.walkValue(child, acc); + if (field === 'value' || field === 'computed_value') values.push(child); else if (field === 'name') names.push(child); - else if (field === 'computed_value') this.walkValue(child, acc); } + // Register result-defs BEFORE the value walk so the call the walk reaches + // carries them — single-name `let x = f()` only (a destructuring or + // multi-binder `let (a, b) = …` / `let p = 1, q = 2` attaches nothing). + const single = this.singlePatternBinder(names); + if (single && values.length === 1) this.registerResultDefs(values[0], [single]); + for (const value of values) this.walkValue(value, acc); for (const pat of names) this.defPattern(pat, acc); return; } @@ -480,12 +530,16 @@ export class SwiftHarvester { const target = node.childForFieldName('target'); const result = node.childForFieldName('result'); const op = node.childForFieldName('operator')?.text ?? '='; + const lv = target ? this.unwrapAssignable(target) : undefined; + const scalar = lv && lv.type === 'simple_identifier' ? lv : undefined; + // A plain `x = ` attaches `resultDefs: [x]`; a compound `+=` does + // not (the prior value flows in too). + if (scalar && op === '=' && result) this.registerResultDefs(result, [scalar]); if (result) this.walkValue(result, acc); - if (target) { - const lv = this.unwrapAssignable(target); - if (lv.type === 'simple_identifier') { - this.def(lv, acc); - if (op !== '=') this.use(lv, acc); // compound assign reads too + if (lv) { + if (scalar) { + this.def(scalar, acc); + if (op !== '=') this.use(scalar, acc); // compound assign reads too } else { // `self.x = …`, `a[i] = …` — root is a use only (not a scalar def). this.walkValue(lv, acc); @@ -493,11 +547,18 @@ export class SwiftHarvester { } return; } + case 'call_expression': + // A Swift call (`foo(a)`, `obj.method(x)`, `a.b.c()`, `Foo(...)`). Records + // a taint site (callee path, receiver, per-arg occurrences, result defs) + // while reproducing the uses the old default descent recorded. Swift has + // no `new` — every site is `kind: 'call'`. + this.visitCall(node, acc); + return; case 'navigation_expression': { - // `a.b` — value read of the chain root only; the suffix name is not a - // scalar binding. - const target = node.childForFieldName('target'); - if (target) this.walkValue(target, acc); + // `a.b` value read — the chain-root identifier is a use plus at most one + // member-read site (the innermost access); the suffix name is not a scalar + // binding. Mirrors the Kotlin / Go value-position member-read semantics. + this.walkChain(node, acc, false); return; } case 'try_expression': { @@ -548,4 +609,145 @@ export class SwiftHarvester { } return n; } + + // ── taint-site harvest ─────────────────────────────────────────────────── + + /** + * The sole `bound_identifier` binder of a single `name` pattern, or undefined + * when there are multiple `name` patterns or the pattern is a tuple / wildcard + * destructuring (`let (a, b) = …`). Used to gate single-target result-defs. + */ + private singlePatternBinder(names: readonly SyntaxNode[]): SyntaxNode | undefined { + if (names.length !== 1) return undefined; + const pat = names[0]; + const bound = pat.childForFieldName?.('bound_identifier'); + if (bound && bound.type === 'simple_identifier' && bound.text !== '_') return bound; + if (pat.type === 'simple_identifier' && pat.text !== '_') return pat; + return undefined; + } + + /** + * When `value`'s root (after unwrapping) is a `call_expression`, remember that + * call site should carry `resultDefs` — the binding indices of `targets` + * (def-position identifiers). Consumed by {@link visitCall} once the value walk + * reaches the node. Single-target only; the blank target (`_`) binds nothing. + */ + private registerResultDefs(value: SyntaxNode, targets: readonly SyntaxNode[]): void { + const root = this.unwrapAssignable(value); + if (root.type !== 'call_expression') return; + const defs: number[] = []; + for (const target of targets) { + if (target.type !== 'simple_identifier' || target.text === '_') continue; + defs.push(this.resolve(target)); + } + if (defs.length > 0) this.resultDefTargets.set(root.id, defs); + } + + /** + * The callee node of a `call_expression` — the first named child that is NOT + * the trailing `call_suffix` (a bare `simple_identifier` for a free / init + * call, or a `navigation_expression` for a member / chained call). + */ + private calleeOf(call: SyntaxNode): SyntaxNode | undefined { + for (let i = 0; i < call.namedChildCount; i++) { + const c = call.namedChild(i); + if (c && c.type !== 'call_suffix') return c; + } + return undefined; + } + + /** + * Open + populate a call site for a Swift `call_expression`. `node` IS the + * `call_expression` — the SAME node the scope query anchors `@reference.call.*` + * on (its `atRange`), so the resolved-id join lands by exact position (see file + * header ANCHOR ALIGNMENT). Swift has no `new`, so every site is `kind: 'call'`. + */ + private visitCall(node: SyntaxNode, acc: FactAccumulator): void { + const calleeNode = this.calleeOf(node); + const siteIdx = acc.openCallSite('call', [ + node.startPosition.row + 1, + node.startPosition.column, + ]); + acc.pushFrame(siteIdx); + if (calleeNode) { + const callee = this.unwrapAssignable(calleeNode); + if (callee.type === 'simple_identifier') { + // A bare free / init call — the callee NAME is a statement-level use but + // NOT a value occurrence in any enclosing argument. + if (callee.text !== '_') acc.addUseWithoutOccurrence(this.resolve(callee)); + acc.setSiteCallee(siteIdx, callee.text); + } else if (callee.type === 'navigation_expression') { + // skipFinalRead: the final `.name` IS the callee, carried by the path. + const chain = this.walkChain(callee, acc, true); + if (chain.path !== undefined) acc.setSiteCallee(siteIdx, chain.path); + if (chain.rootIdx !== undefined) acc.setSiteReceiver(siteIdx, chain.rootIdx); + } else { + // Call-rooted chains (`f()()`), parenthesized callables, `self.x`-rooted — + // the walk still records uses and nested sites; no static callee path. + this.walkValue(callee, acc); + } + } + const resultDefs = this.resultDefTargets.get(node.id); + if (resultDefs !== undefined) acc.setSiteResultDefs(siteIdx, resultDefs); + const suffix = node.namedChildren.find((c) => c.type === 'call_suffix'); + if (suffix) this.walkArguments(suffix, acc); + acc.popFrame(); + } + + /** + * Walk a `call_suffix`'s `value_arguments`, tagging each positional / labeled + * argument's occurrence position. A trailing closure (`lambda_literal`) is a + * nested function body — opaque (excluded by {@link NESTED_FUNCTION_TYPES}), so + * it is not an argument occurrence here. + */ + private walkArguments(suffix: SyntaxNode, acc: FactAccumulator): void { + const args = suffix.namedChildren.find((c) => c.type === 'value_arguments'); + if (!args) return; + let pos = 0; + for (let i = 0; i < args.namedChildCount; i++) { + const arg = args.namedChild(i); + if (!arg || arg.type !== 'value_argument') continue; + acc.setFrameArg(pos); + // A labeled arg (`name: value`) records only the VALUE occurrence — the + // `value_argument_label` is dropped. An `&inout` value walks its `target` + // identifier for the use. Swift has no call-site spread operator. + const value = arg.childForFieldName('value'); + if (value) this.walkValue(value, acc); + pos++; + } + } + + /** + * `navigation_expression` chain walk shared by value position and callee + * position. Records the chain-root identifier as a use plus at most ONE + * member-read site — the INNERMOST access — when the root is an identifier; + * `skipFinalRead` suppresses it when that access is the callee (carried by the + * dotted path instead). Mirrors the Kotlin / Go / Python harvesters' walkChain. + * A non-identifier root (`self`/literal/call) launders no static path/receiver. + */ + private walkChain( + node: SyntaxNode, + acc: FactAccumulator, + skipFinalRead: boolean, + ): { path?: string; rootIdx?: number } { + const accesses: string[] = []; + let cur: SyntaxNode = this.unwrapAssignable(node); + for (;;) { + if (cur.type === 'navigation_expression') { + const suffix = cur.childForFieldName('suffix'); + const name = suffix?.childForFieldName('suffix'); + accesses.unshift(name?.text ?? ''); + const operand = cur.childForFieldName('target'); + if (!operand) break; + cur = this.unwrapAssignable(operand); + } else { + break; + } + } + // The shared terminal: root-use record + innermost member-read + path-join. + return finalizeChain(acc, cur, accesses, skipFinalRead, (t) => t === 'simple_identifier', { + resolve: (n) => this.resolve(n), + walkRoot: (n) => this.walkValue(n, acc), + }); + } } diff --git a/gitnexus/src/core/ingestion/cfg/visitors/typescript-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/typescript-harvest.ts index f1ea3abc0..82fdd9d09 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/typescript-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/typescript-harvest.ts @@ -186,6 +186,7 @@ export class TsHarvester { kind: BindingEntry['kind'], scope: Scope, hoistToRoot: boolean, + formalIndex?: number, ): void { const target = hoistToRoot ? this.root : scope; const name = nameNode.text; @@ -200,6 +201,10 @@ export class TsHarvester { declLine: nameNode.startPosition.row + 1, declColumn: nameNode.startPosition.column, kind, + // Carry the enclosing formal position for params so the PDG call-summary + // consumer joins by FORMAL slot, never the flattened binding ordinal. A + // destructured/rest formal hands the SAME index to every inner name. + ...(formalIndex !== undefined ? { formalIndex } : {}), }); } @@ -207,48 +212,60 @@ export class TsHarvester { const params = fnNode.childForFieldName('parameters') ?? fnNode.childForFieldName('parameter'); if (!params) return; if (params.type === 'identifier') { - this.declare(params, 'param', this.root, true); // `x => …` single-param arrow + this.declare(params, 'param', this.root, true, 0); // `x => …` single-param arrow ⇒ formal 0 return; } + // Each NAMED child of the parameter list is one top-level formal — its index + // here is the 0-based formal position. Destructured/rest formals fan out to + // several inner names, but ALL inherit this one formal index, so the PDG + // call-summary never misattributes an inner name to a later formal's slot. + let formalIndex = 0; for (let i = 0; i < params.namedChildCount; i++) { const p = params.namedChild(i); if (!p) continue; // TS wraps each param (required_parameter/optional_parameter, field // `pattern`); plain JS puts the pattern directly in formal_parameters. const pattern = p.childForFieldName('pattern') ?? p; - this.declarePattern(pattern, 'param', this.root, true); + this.declarePattern(pattern, 'param', this.root, true, formalIndex); + formalIndex++; } } - /** Declare every name bound by a (possibly destructuring) pattern. */ + /** + * Declare every name bound by a (possibly destructuring) pattern. When + * `formalIndex` is supplied (param patterns), EVERY name the pattern binds + * carries that one enclosing-formal position (the recursion never reassigns + * it), so `function f({a, b}, c)` records a:0, b:0, c:1. + */ private declarePattern( node: SyntaxNode, kind: BindingEntry['kind'], scope: Scope, hoistToRoot: boolean, + formalIndex?: number, ): void { switch (node.type) { case 'identifier': case 'shorthand_property_identifier_pattern': - this.declare(node, kind, scope, hoistToRoot); + this.declare(node, kind, scope, hoistToRoot, formalIndex); return; case 'rest_pattern': case 'object_pattern': case 'array_pattern': for (let i = 0; i < node.namedChildCount; i++) { const c = node.namedChild(i); - if (c) this.declarePattern(c, kind, scope, hoistToRoot); + if (c) this.declarePattern(c, kind, scope, hoistToRoot, formalIndex); } return; case 'pair_pattern': { const value = node.childForFieldName('value'); - if (value) this.declarePattern(value, kind, scope, hoistToRoot); + if (value) this.declarePattern(value, kind, scope, hoistToRoot, formalIndex); return; } case 'assignment_pattern': case 'object_assignment_pattern': { const left = node.childForFieldName('left'); - if (left) this.declarePattern(left, kind, scope, hoistToRoot); + if (left) this.declarePattern(left, kind, scope, hoistToRoot, formalIndex); return; } default: @@ -256,7 +273,7 @@ export class TsHarvester { for (let i = 0; i < node.namedChildCount; i++) { const c = node.namedChild(i); if (c && !TYPE_CONTEXT_TYPES.has(c.type)) { - this.declarePattern(c, kind, scope, hoistToRoot); + this.declarePattern(c, kind, scope, hoistToRoot, formalIndex); } } } @@ -716,7 +733,9 @@ export class TsHarvester { private visitCall(node: SyntaxNode, acc: FactAccumulator, kind: 'call' | 'new'): void { const calleeNode = node.childForFieldName(kind === 'new' ? 'constructor' : 'function'); const argsNode = node.childForFieldName('arguments'); - const siteIdx = acc.openCallSite(kind); + // `node` IS the call_expression/new_expression — the SAME node the + // scope-extractor anchors `@reference.call.*` (its `atRange`) on (KTD7). + const siteIdx = acc.openCallSite(kind, [node.startPosition.row + 1, node.startPosition.column]); acc.pushFrame(siteIdx); let calleePath: string | undefined; if (calleeNode) { @@ -862,6 +881,8 @@ interface MutableSite { requireArg?: string; object?: number; property?: string; + /** Call-site anchor position — see {@link SiteRecord.at}. Call/new only. */ + at?: [number, number]; } /** @@ -945,11 +966,17 @@ class FactAccumulator { return [...this.defs.slice(snap[0]), ...this.mayDefs.slice(snap[1])]; } - /** Open a call/new site; parent = innermost enclosing argument position. */ - openCallSite(kind: 'call' | 'new'): number { + /** + * Open a call/new site; parent = innermost enclosing argument position. `at` + * is the call/new node's anchor position `[line (1-based), col (0-based)]` — + * the SAME position the CALLS-edge resolution keys on (KTD7; see + * {@link SiteRecord.at}). + */ + openCallSite(kind: 'call' | 'new', at?: readonly [number, number]): number { const site: MutableSite = { kind }; const parent = this.innermostArgPosition(); if (parent) site.parent = parent; + if (at) site.at = [at[0], at[1]]; this.sites.push(site); return this.sites.length - 1; } diff --git a/gitnexus/src/core/ingestion/pipeline-phases/call-summaries.ts b/gitnexus/src/core/ingestion/pipeline-phases/call-summaries.ts new file mode 100644 index 000000000..6b1db0aa7 --- /dev/null +++ b/gitnexus/src/core/ingestion/pipeline-phases/call-summaries.ts @@ -0,0 +1,73 @@ +/** + * Phase: callSummaries (PDG FU-C, U-C3) + * + * The whole-program CALL_SUMMARY materialisation pass — the dependence-engine + * SIBLING of `taintSummaries`. Runs AFTER scope-resolution (where the resolved + * `CALLS` graph lives in `ctx.graph` and the per-function RETURN-VALUE ASCENT + * summaries were harvested in-phase) and emits one `CALL_SUMMARY` self-loop edge + * per harvested callee. A later consumer phase (NOT this task) decodes the + * bitset to ascend a callee's return effect into the caller continuation. + * + * Opt-in: registered with `enabledWhen: (o) => o.pdg === true`. A default + * `analyze` run never includes it, so the graph is byte-identical and emits ZERO + * CALL_SUMMARY edges. No always-on phase depends on it. + * + * @deps scopeResolution, pruneLocalSymbols + * @reads scopeResolution output (callSummaries) + * @writes graph (CALL_SUMMARY self-loop edges) + */ + +import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js'; +import { getPhaseOutput } from './types.js'; +import type { ScopeResolutionOutput } from '../scope-resolution/pipeline/phase.js'; +import { + emitCallSummaries, + DEFAULT_PDG_MAX_CALL_SUMMARY_EDGES, +} from '../taint/call-summary-emit.js'; +import { logger } from '../../logger.js'; + +export interface CallSummariesOutput { + /** Per-callee summaries fed to the emit. */ + summaries: number; + /** CALL_SUMMARY edges persisted. */ + edgesEmitted: number; +} + +const EMPTY: CallSummariesOutput = { summaries: 0, edgesEmitted: 0 }; + +export const callSummariesPhase: PipelinePhase = { + name: 'callSummaries', + deps: ['scopeResolution', 'pruneLocalSymbols'], + + async execute( + ctx: PipelineContext, + deps: ReadonlyMap>, + ): Promise { + const scope = getPhaseOutput(deps, 'scopeResolution'); + const summaries = scope.callSummaries; + if (summaries.length === 0) return EMPTY; + + const maxEdges = ctx.options?.pdgMaxCallSummaryEdges ?? DEFAULT_PDG_MAX_CALL_SUMMARY_EDGES; + const emit = emitCallSummaries(ctx.graph, summaries, { maxEdges }, (m) => logger.warn(m)); + + if (emit.edgesDropped > 0) { + logger.warn( + `[call-summary] capped: ${emit.edgesDropped} CALL_SUMMARY edge(s) dropped by the ` + + `per-run cap (${maxEdges}) — raise pdgMaxCallSummaryEdges if intentional`, + ); + } + if (emit.skippedMissingEndpoint > 0) { + logger.debug( + `[call-summary] ${emit.skippedMissingEndpoint} summary/summaries skipped (callee node ` + + `missing from graph)`, + ); + } + if (emit.edgesEmitted > 0) { + logger.debug( + `[call-summary] ${summaries.length} summaries → ${emit.edgesEmitted} CALL_SUMMARY edge(s)`, + ); + } + + return { summaries: summaries.length, edgesEmitted: emit.edgesEmitted }; + }, +}; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/index.ts b/gitnexus/src/core/ingestion/pipeline-phases/index.ts index fa458b6e2..15b5e6777 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/index.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/index.ts @@ -22,6 +22,7 @@ export { } from '../scope-resolution/pipeline/phase.js'; export { pruneLocalSymbolsPhase, type PruneLocalSymbolsOutput } from './prune-local-symbols.js'; export { taintSummariesPhase, type TaintSummariesOutput } from './taint-summaries.js'; +export { callSummariesPhase, type CallSummariesOutput } from './call-summaries.js'; export { mroPhase, type MROOutput } from './mro.js'; export { communitiesPhase, type CommunitiesOutput } from './communities.js'; export { processesPhase, type ProcessesOutput } from './processes.js'; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/routes.ts b/gitnexus/src/core/ingestion/pipeline-phases/routes.ts index 561590e05..e9909afbe 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/routes.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/routes.ts @@ -42,6 +42,15 @@ const EXPO_NAV_PATTERNS = [ export interface RouteEntry { filePath: string; source: string; + /** + * HTTP verb for this route when ingestion knows it structurally + * (Spring/Laravel framework routes and decorator routes carry + * `httpMethod`; filesystem-derived routes — Next.js/Expo/PHP file + * routes — do not, so this stays undefined for them). Persisted onto + * the Route node so downstream contract extraction can read the verb + * from the graph instead of re-parsing the handler source. + */ + method?: string; } export interface RoutesOutput { @@ -135,6 +144,33 @@ function escapeRegex(s: string): string { return s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); } +/** + * Canonicalize a route's HTTP verb for persistence on the Route node. + * Returns an upper-cased standard method, or `undefined` when the value + * is not a real HTTP verb. Laravel `Route::resource` / `apiResource` + * surface `httpMethod` values like `resource` / `apiResource` (they + * expand to several verbs at runtime), so they must not be stored as a + * method — leaving them `undefined` keeps the column clean and lets the + * contract extractor fall back to its source-scan path for those routes. + */ +const VALID_HTTP_METHODS = new Set([ + 'GET', + 'POST', + 'PUT', + 'PATCH', + 'DELETE', + 'HEAD', + 'OPTIONS', + 'TRACE', + 'CONNECT', +]); + +export function normalizeRouteMethod(raw: string | null | undefined): string | undefined { + if (typeof raw !== 'string') return undefined; + const verb = raw.trim().toUpperCase(); + return VALID_HTTP_METHODS.has(verb) ? verb : undefined; +} + export const routesPhase: PipelinePhase = { name: 'routes', deps: ['parse'], @@ -213,6 +249,7 @@ export const routesPhase: PipelinePhase = { addRoute(routeUrl, { filePath: route.filePath, source: 'framework-route', + method: normalizeRouteMethod(route.httpMethod), }); if (route.routeName && !namedRouteRegistry.has(route.routeName)) { namedRouteRegistry.set(route.routeName, routeUrl); @@ -223,6 +260,7 @@ export const routesPhase: PipelinePhase = { addRoute(url, { filePath: dr.filePath, source: `decorator-${dr.decoratorName}`, + method: normalizeRouteMethod(dr.httpMethod), }); } @@ -232,7 +270,7 @@ export const routesPhase: PipelinePhase = { handlerContents = await readFileContents(ctx.repoPath, handlerPaths); for (const [routeURL, entry] of routeRegistry) { - const { filePath: handlerPath, source: routeSource } = entry; + const { filePath: handlerPath, source: routeSource, method: routeMethod } = entry; const content = handlerContents.get(handlerPath); const { responseKeys, errorKeys } = content @@ -251,6 +289,7 @@ export const routesPhase: PipelinePhase = { properties: { name: routeURL, filePath: handlerPath, + ...(routeMethod ? { method: routeMethod } : {}), ...(responseKeys ? { responseKeys } : {}), ...(errorKeys ? { errorKeys } : {}), ...(middleware && middleware.length > 0 ? { middleware } : {}), diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index ff01ccdb6..9c27a5250 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -33,6 +33,7 @@ import { scopeResolutionPhase, pruneLocalSymbolsPhase, taintSummariesPhase, + callSummariesPhase, mroPhase, communitiesPhase, processesPhase, @@ -121,6 +122,11 @@ export interface PipelineOptions { /** Per-run `TAINT_PATH` edge cap (#2084 review P1-3). `undefined` ⇒ * `DEFAULT_PDG_MAX_INTERPROC_EDGES` (1000); `0` ⇒ no cap. */ pdgMaxInterprocEdges?: number; + /** Per-run `CALL_SUMMARY` edge cap (PDG FU-C, U-C3). `undefined` ⇒ + * `DEFAULT_PDG_MAX_CALL_SUMMARY_EDGES` (0 = unlimited); `0` ⇒ no cap. + * Programmatic only, no CLI flag (KTD8) — same discipline as the other + * pdg caps. */ + pdgMaxCallSummaryEdges?: number; /** * Streaming/chunked PDG graph emit (#2202). When true, the BasicBlock + * intra-file PDG-edge layer (CFG / REACHING_DEF / CDG / POST_DOMINATE / @@ -267,6 +273,7 @@ export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] { // pdg-gated phase. Off ⇒ absent ⇒ byte-identical graph. No always-on // phase depends on it (a filtered-out dep would throw in getPhaseOutput). .register(taintSummariesPhase, { enabledWhen: (o) => o.pdg === true }) + .register(callSummariesPhase, { enabledWhen: (o) => o.pdg === true }) .register(mroPhase, { enabledWhen: (o) => !o.skipGraphPhases }) .register(communitiesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) .register(processesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) diff --git a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/callee-id-sink.ts b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/callee-id-sink.ts new file mode 100644 index 000000000..af878b3b1 --- /dev/null +++ b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/callee-id-sink.ts @@ -0,0 +1,106 @@ +/** + * Resolved-callee-id capture sink (#2227 follow-up plan U2). + * + * During Phase-4 scope-resolution CALLS-edge emission, each resolved call + * site's `(callSiteLine, callSiteCol) → resolvedCalleeId` mapping is + * accumulated here — across ALL THREE CALLS emit paths + * (`emitReceiverBoundCalls` via `tryEmitEdge`/`tryEmitEdgeWithExplicitTargetId`, + * and the inline `graph.addRelationship` in `emitFreeCallFallback` and + * `emitReferencesViaLookup`), each BEFORE its dedup (KTD6/R8). A later unit + * (U3) joins this map to CFG `BasicBlock`s by exact call-site position and + * emits a `BasicBlock.calleeIds` set. + * + * KEY ALIGNMENT (plan KTD7 — load-bearing): the key is the call/new + * expression node's start position — `line` 1-based (`startPosition.row + 1`), + * `col` 0-based (`startPosition.column`). This MUST equal the U1 + * `SiteRecord.at` so the U3 position join lands. The CALLS resolution exposes + * the same node's range via `site.atRange` (`atRange: anchor.range`, + * scope-extractor.ts:1030), whose `startLine`/`startCol` are built by + * `nodeToCapture` as `row + 1` / `column` (1-based line, 0-based col — see the + * `Range` doc in gitnexus-shared). So a capture keyed on + * `(atRange.startLine, atRange.startCol)` is byte-equal to U1's `at` — no + * normalization needed. + * + * Gating (R4): the concrete sink is created in `run.ts` only when + * `input.pdg === true`; otherwise `undefined` is threaded through, so off-mode + * does zero work and emits byte-identical output. + * + * Multi-target dispatch (R2/KTD8): one site → multiple emit calls → the `Set` + * accumulates every resolved target. Capture is per-emit-call, so the + * candidate set is complete and a real target is never dropped. + */ + +/** Encoded position key: `${line}:${col}` (1-based line, 0-based col). */ +export type CalleeIdPosKey = string; + +/** Build the position key from a call-site anchor. Single source of truth so + * producer (this sink) and consumer (U3's CFG join) encode positions + * identically. */ +export function calleeIdPosKey(line: number, col: number): CalleeIdPosKey { + return `${line}:${col}`; +} + +/** + * Write-side contract handed to the three CALLS emitters. Each resolved CALLS + * edge feeds one `add` BEFORE its dedup, keyed on the call-site anchor. + */ +export interface CalleeIdSink { + /** + * Record that the call site at `(line, col)` in `filePath` resolved to + * `calleeId`. Idempotent per `(filePath, line, col, calleeId)` — the + * underlying value is a `Set`, so repeat targets collapse. + */ + add(filePath: string, line: number, col: number, calleeId: string): void; +} + +/** + * Read-side accessor consumed by U3's CFG-emit join. Returns the per-file + * `posKey → Set` map (or `undefined` when the file produced no + * captures). Kept separate from the write interface so the emitters only see + * the narrow `add` surface. + */ +export interface CalleeIdMapView { + /** Per-file position→ids map, or `undefined` if nothing was captured for it. */ + get(filePath: string): ReadonlyMap> | undefined; + /** + * Release a file's captured map once its CFG emit has consumed it (R6). The + * three CALLS passes fully precede the CFG-emit loop and each file is read + * exactly once, so releasing after consumption bounds the accumulator to one + * file's call sites instead of holding the whole repo's for the full phase. + */ + delete(filePath: string): void; +} + +/** The concrete accumulator: a write sink that also exposes the read view. */ +export interface CalleeIdAccumulator extends CalleeIdSink, CalleeIdMapView {} + +/** + * Create the concrete nested-`Map` accumulator. Call ONLY when + * `input.pdg === true` (else thread `undefined` for byte-identity / zero + * overhead — R4). + */ +export function createCalleeIdAccumulator(): CalleeIdAccumulator { + const byFile = new Map>>(); + return { + add(filePath: string, line: number, col: number, calleeId: string): void { + let byPos = byFile.get(filePath); + if (byPos === undefined) { + byPos = new Map>(); + byFile.set(filePath, byPos); + } + const key = calleeIdPosKey(line, col); + let ids = byPos.get(key); + if (ids === undefined) { + ids = new Set(); + byPos.set(key, ids); + } + ids.add(calleeId); + }, + get(filePath: string): ReadonlyMap> | undefined { + return byFile.get(filePath); + }, + delete(filePath: string): void { + byFile.delete(filePath); + }, + }; +} diff --git a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/edges.ts b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/edges.ts index ce7b52844..6aa72fc25 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/edges.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/edges.ts @@ -20,6 +20,17 @@ import type { KnowledgeGraph } from '../../../graph/types.js'; import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; import type { GraphNodeLookup } from '../graph-bridge/node-lookup.js'; import { resolveCallerGraphId, resolveDefGraphId } from '../graph-bridge/ids.js'; +import type { CalleeIdSink } from './callee-id-sink.js'; + +/** + * Optional resolved-callee-id capture context (#2227 follow-up U2). Threaded + * in only under `--pdg` (else `undefined` → zero overhead, byte-identity R4). + * `filePath` is NOT on the `site` param, so it rides here alongside the sink. + */ +export interface CalleeIdCaptureCtx { + readonly sink: CalleeIdSink; + readonly filePath: string; +} /** * Map a `Reference.kind` to a graph edge type. `import-use` is dropped @@ -74,6 +85,7 @@ export function tryEmitEdge( seen: Set, confidence = 0.85, collapseByCallerTarget = false, + calleeCapture?: CalleeIdCaptureCtx, ): boolean { // Inheritance edges are emitted directly by `preEmitInheritanceEdges` (which // owns the enclosing-class caller and the EXTENDS-vs-IMPLEMENTS type), so this @@ -85,6 +97,19 @@ export function tryEmitEdge( if (targetGraphId === undefined) return false; if (edgeType === undefined) return false; + // Resolved-callee-id capture (#2227 U2/KTD6/R8): record this CALLS site's + // resolved target BEFORE the dedup `seen` check, so collapsed same-target + // multi-line calls are still captured per site. Keyed on `site.atRange` + // (1-based line / 0-based col — byte-equal to U1's SiteRecord.at). + if (calleeCapture !== undefined && edgeType === 'CALLS') { + calleeCapture.sink.add( + calleeCapture.filePath, + site.atRange.startLine, + site.atRange.startCol, + targetGraphId, + ); + } + // CALLS edges may collapse to `(caller, target)` granularity when // the provider opts in (C# matches legacy DAG behavior this way). // Write/read ACCESSES keep per-site dedup so multiple writes to the @@ -134,12 +159,24 @@ export function tryEmitEdgeWithExplicitTargetId( seen: Set, confidence = 0.85, collapseByCallerTarget = false, + calleeCapture?: CalleeIdCaptureCtx, ): boolean { const callerGraphId = resolveCallerGraphId(site.inScope, scopes, nodeLookup, site.atRange); const edgeType = mapReferenceKindToEdgeType(site.kind as Reference['kind']); if (callerGraphId === undefined) return false; if (edgeType === undefined) return false; + // Resolved-callee-id capture (#2227 U2/KTD6/R8) — before dedup, see + // `tryEmitEdge`. The explicit target id IS the resolved callee id. + if (calleeCapture !== undefined && edgeType === 'CALLS') { + calleeCapture.sink.add( + calleeCapture.filePath, + site.atRange.startLine, + site.atRange.startCol, + targetGraphId, + ); + } + const useCollapsed = collapseByCallerTarget && edgeType === 'CALLS'; const dedupKey = useCollapsed ? `${edgeType}:${callerGraphId}->${targetGraphId}` diff --git a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/references-to-edges.ts b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/references-to-edges.ts index eb8bfdfea..c3c301d86 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/references-to-edges.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/references-to-edges.ts @@ -24,6 +24,7 @@ import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexe import { resolveCallerGraphId, resolveDefGraphId } from '../graph-bridge/ids.js'; import { mapReferenceKindToEdgeType } from '../graph-bridge/edges.js'; import type { GraphNodeLookup } from '../graph-bridge/node-lookup.js'; +import type { CalleeIdSink } from '../graph-bridge/callee-id-sink.js'; /** * Optional opaque skip key — providers may pre-emit edges (e.g. via @@ -40,6 +41,10 @@ export function emitReferencesViaLookup( referenceIndex: { readonly bySourceScope: ReadonlyMap }, nodeLookup: GraphNodeLookup, skipSites?: ReferenceSiteSkipSet, + /** Resolved-callee-id capture sink (#2227 U2). Threaded in only under + * `--pdg`; `undefined` ⇒ zero overhead, byte-identity (R4). Captured at the + * CALLS emit below BEFORE this loop's `seen` dedup (KTD6/R8). */ + calleeIdSink?: CalleeIdSink, ): { emitted: number; skipped: number } { let emitted = 0; let skipped = 0; @@ -80,6 +85,15 @@ export function emitReferencesViaLookup( continue; } + // Resolved-callee-id capture (#2227 U2/KTD6/R8): record this CALLS site's + // resolved target BEFORE the `seen` dedup, keyed on `ref.atRange` + // (byte-equal to U1's SiteRecord.at: 1-based line / 0-based col). Only + // CALLS feeds the bridge; ACCESSES/USES/EXTENDS are skipped. `fromFilePath` + // is the call-site (caller) file — the same file U1 stamps the site on. + if (calleeIdSink !== undefined && edgeType === 'CALLS' && fromFilePath !== undefined) { + calleeIdSink.add(fromFilePath, ref.atRange.startLine, ref.atRange.startCol, targetGraphId); + } + const dedupKey = `${edgeType}:${callerGraphId}->${targetGraphId}:${ref.atRange.startLine}:${ref.atRange.startCol}`; if (seen.has(dedupKey)) continue; seen.add(dedupKey); diff --git a/gitnexus/src/core/ingestion/scope-resolution/passes/free-call-fallback.ts b/gitnexus/src/core/ingestion/scope-resolution/passes/free-call-fallback.ts index e680ea4e7..04382f6be 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/passes/free-call-fallback.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/passes/free-call-fallback.ts @@ -35,6 +35,7 @@ import type { ResolutionSuppressionReason, } from '../resolution-outcome.js'; import { resolveCallerGraphId, resolveDefGraphId } from '../graph-bridge/ids.js'; +import type { CalleeIdSink } from '../graph-bridge/callee-id-sink.js'; import { findAllCallableBindingsInScope, findCallableBindingInScope, @@ -88,6 +89,11 @@ export function emitFreeCallFallback( * candidate (monotonicity). */ readonly constraintCompatibility?: ScopeResolver['constraintCompatibility']; readonly recordResolutionOutcome?: ResolutionOutcomeRecorder; + /** Resolved-callee-id capture sink (#2227 U2). Threaded in only under + * `--pdg`; `undefined` ⇒ zero overhead, byte-identity (R4). Captured at + * the CALLS emit below BEFORE the collapsed `seen` dedup (KTD6) so + * same-target multi-line calls are still recorded per site. */ + readonly calleeIdSink?: CalleeIdSink; } = {}, ): number { let emitted = 0; @@ -422,6 +428,18 @@ export function emitFreeCallFallback( // means we don't add a new edge — so `emit-references` skips its // potentially-wrong fallback for the same site. handledSites.add(siteKey(parsed.filePath, site)); + // Resolved-callee-id capture (#2227 U2/KTD6/R8): record this CALLS site's + // resolved target BEFORE the collapsed `seen` dedup. The free-call dedup + // key drops the line (one edge per caller→target), so capturing after + // `seen.has` would lose every same-target call past the first — capture + // here, per site, keyed on `site.atRange` (byte-equal to U1's + // SiteRecord.at: 1-based line / 0-based col). + options.calleeIdSink?.add( + parsed.filePath, + site.atRange.startLine, + site.atRange.startCol, + tgtGraphId, + ); const relId = `rel:CALLS:${callerGraphId}->${tgtGraphId}`; if (seen.has(relId)) continue; seen.add(relId); diff --git a/gitnexus/src/core/ingestion/scope-resolution/passes/receiver-bound-calls.ts b/gitnexus/src/core/ingestion/scope-resolution/passes/receiver-bound-calls.ts index 88139e3c1..21eef7aaa 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/passes/receiver-bound-calls.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/passes/receiver-bound-calls.ts @@ -54,7 +54,12 @@ import { findValueBindingInScope, isClassLike, } from '../scope/walkers.js'; -import { tryEmitEdge, tryEmitEdgeWithExplicitTargetId } from '../graph-bridge/edges.js'; +import { + tryEmitEdge, + tryEmitEdgeWithExplicitTargetId, + type CalleeIdCaptureCtx, +} from '../graph-bridge/edges.js'; +import type { CalleeIdSink } from '../graph-bridge/callee-id-sink.js'; import { resolveCompoundReceiverClass } from '../passes/compound-receiver.js'; import { resolveDefGraphId } from '../graph-bridge/ids.js'; import { @@ -148,6 +153,10 @@ export function emitReceiverBoundCalls( model: SemanticModel, options: { readonly recordResolutionOutcome?: ResolutionOutcomeRecorder; + /** Resolved-callee-id capture sink (#2227 U2). Threaded in only under + * `--pdg`; `undefined` ⇒ zero overhead, byte-identity (R4). Per-file + * capture contexts are built from this + `parsed.filePath` in the loop. */ + readonly calleeIdSink?: CalleeIdSink; } = {}, ): number { let emitted = 0; @@ -200,6 +209,7 @@ export function emitReceiverBoundCalls( primaryMemberDef: SymbolDefinition, site: ParsedFile['referenceSites'][number], confidence: number, + calleeCapture: CalleeIdCaptureCtx | undefined, ): number => { if (ownerDef.type !== 'Interface') return 0; const impls = implementorsByInterfaceDefId.get(ownerDef.nodeId); @@ -225,6 +235,7 @@ export function emitReceiverBoundCalls( seen, confidence, collapse, + calleeCapture, ); if (ok) n++; } @@ -233,6 +244,13 @@ export function emitReceiverBoundCalls( for (const parsed of parsedFiles) { const namespaceTargets = collectNamespaceTargets(parsed, scopes); + // Per-file resolved-callee-id capture context (#2227 U2). Built once per + // file; `undefined` when the sink is absent (pdg off) so the `tryEmitEdge` + // capture is a no-op and emission stays byte-identical (R4). + const calleeCapture: CalleeIdCaptureCtx | undefined = + options.calleeIdSink !== undefined + ? { sink: options.calleeIdSink, filePath: parsed.filePath } + : undefined; for (const site of parsed.referenceSites) { if (site.kind !== 'call' && site.kind !== 'read' && site.kind !== 'write') continue; @@ -330,6 +348,7 @@ export function emitReceiverBoundCalls( seen, 0.85, collapse, + calleeCapture, ); if (ok) emitted++; // Always mark handled when the site was resolved, even @@ -417,6 +436,7 @@ export function emitReceiverBoundCalls( seen, 0.85, collapse, + calleeCapture, ); if (ok) emitted++; // Always mark handled when the site was resolved, even @@ -497,6 +517,7 @@ export function emitReceiverBoundCalls( seen, confidence, collapse, + calleeCapture, ); if (ok) emitted++; handledSites.add(siteKey); @@ -593,6 +614,7 @@ export function emitReceiverBoundCalls( seen, confidence, collapse, + calleeCapture, ); if (ok) emitted++; handledSites.add(siteKey); @@ -630,6 +652,7 @@ export function emitReceiverBoundCalls( seen, 0.85, collapse, + calleeCapture, ); if (ok) emitted++; handledSites.add(siteKey); @@ -692,6 +715,7 @@ export function emitReceiverBoundCalls( seen, 0.85, collapse, + calleeCapture, ); if (ok) emitted++; handledSites.add(siteKey); @@ -773,6 +797,7 @@ export function emitReceiverBoundCalls( seen, confidence, collapse, + calleeCapture, ); if (ok) emitted++; handledSites.add(siteKey); @@ -831,6 +856,11 @@ export function emitReceiverBoundCalls( memberDef, memberDef.filePath !== parsed.filePath ? 'import-resolved' : 'global', seen, + // Explicit defaults so the trailing capture ctx (#2227 U2) can + // be threaded without changing dedup/confidence behavior. + 0.85, + false, + calleeCapture, ); if (ok) { emitted++; @@ -939,6 +969,7 @@ export function emitReceiverBoundCalls( seen, 0.85, collapse, + calleeCapture, ); if (ok) emitted++; // Always mark handled when the site was resolved, even @@ -1029,6 +1060,7 @@ export function emitReceiverBoundCalls( seen, confidence, collapse, + calleeCapture, ); if (ok) emitted++; handledSites.add(siteKey); @@ -1145,12 +1177,20 @@ export function emitReceiverBoundCalls( seen, confidence, collapse, + calleeCapture, ); if (ok) emitted++; // Interface dispatch: when the primary owner is an // Interface, emit secondary CALLS edges to every // implementing class's same-named method. - emitted += emitInterfaceDispatchFor(ownerDef, memberName, memberDef, site, confidence); + emitted += emitInterfaceDispatchFor( + ownerDef, + memberName, + memberDef, + site, + confidence, + calleeCapture, + ); // Always mark handled when the site was resolved, even // if the edge was deduplicated (collapse mode), so // `emitReferencesViaLookup` doesn't re-emit from the @@ -1242,6 +1282,7 @@ export function emitReceiverBoundCalls( seen, confidence, collapse, + calleeCapture, ); if (ok) emitted++; handledSites.add(siteKey); diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index c7671d8f4..f88cbbaaa 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -43,6 +43,7 @@ import { } from '../../../../storage/parsedfile-store.js'; import type { ResolutionOutcome } from '../resolution-outcome.js'; import type { FunctionSummary } from '../../taint/summary-model.js'; +import type { CallSummary } from '../../taint/call-summary-model.js'; import { buildFunctionNodeIndex } from '../../taint/summary-harvest-driver.js'; import { PdgEmitSink, type PdgEmitManifest } from '../../../lbug/pdg-emit-sink.js'; import { resolveNativeSafeStorageDir } from '../../../lbug/lbug-config.js'; @@ -74,6 +75,13 @@ export interface ScopeResolutionOutput { * The `taintSummaries` phase composes these over the `CALLS` graph. */ readonly functionSummaries: readonly FunctionSummary[]; + /** + * Per-function RETURN-VALUE ASCENT summaries harvested in the pdg window + * (PDG FU-C, U-C2), across all languages. Empty unless `--pdg`. The + * `callSummaries` phase materialises one `CALL_SUMMARY` self-loop edge per + * entry once the resolved call graph is known. + */ + readonly callSummaries: readonly CallSummary[]; /** * Streamed PDG-emit COPY manifest (#2202). Present only when streaming was on * (full rebuild + `--pdg` + enabled): the BasicBlock node CSV + per-pair PDG @@ -92,6 +100,7 @@ const NOOP_OUTPUT: ScopeResolutionOutput = Object.freeze({ resolutionOutcomes: [], perLanguage: new Map(), functionSummaries: [], + callSummaries: [], }); export const scopeResolutionPhase: PipelinePhase = { @@ -165,6 +174,9 @@ export const scopeResolutionPhase: PipelinePhase = { // M4 (#2084 U1): per-function taint summaries accumulated across every // language pass; the cross-function fixpoint phase reads this output. const functionSummaries: FunctionSummary[] = []; + // FU-C (U-C2): per-function RETURN-VALUE ASCENT summaries accumulated across + // every language pass; the `callSummaries` emit phase reads this output. + const callSummaries: CallSummary[] = []; const perLanguage = new Map< SupportedLanguages, { @@ -510,6 +522,7 @@ export const scopeResolutionPhase: PipelinePhase = { processedScopeFiles += langFileCount; anyRan = true; functionSummaries.push(...stats.functionSummaries); + callSummaries.push(...stats.callSummaries); totalFiles += stats.filesProcessed; totalImports += stats.importsEmitted; totalRefs += stats.referenceEdgesEmitted; @@ -572,6 +585,7 @@ export const scopeResolutionPhase: PipelinePhase = { resolutionOutcomes, perLanguage, functionSummaries, + callSummaries, pdgEmitManifest, }; }, diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts index 8c9222325..e8d74d194 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts @@ -45,6 +45,7 @@ import { DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, REACHING_DEF_FACTS_PER_EDGE_CAP, } from '../../cfg/emit.js'; +import { createMemoizedReachingDefs } from '../../cfg/reaching-defs.js'; import { emitFileTaint, DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, @@ -58,7 +59,9 @@ import { harvestFileSummaries, type FunctionNodeIndex, } from '../../taint/summary-harvest-driver.js'; +import { harvestFileCallSummaries } from '../../taint/summary-harvest-driver.js'; import type { FunctionSummary } from '../../taint/summary-model.js'; +import type { CallSummary } from '../../taint/call-summary-model.js'; import type { FunctionCfg } from '../../cfg/types.js'; import { resolveDefGraphId } from '../graph-bridge/ids.js'; import { buildPopulatedMethodDispatch } from '../graph-bridge/method-dispatch.js'; @@ -66,6 +69,10 @@ import { propagateImportedReturnTypes } from '../passes/imported-return-types.js import { emitReceiverBoundCalls } from '../passes/receiver-bound-calls.js'; import { emitFreeCallFallback } from '../passes/free-call-fallback.js'; import { emitReferencesViaLookup } from '../graph-bridge/references-to-edges.js'; +import { + createCalleeIdAccumulator, + type CalleeIdAccumulator, +} from '../graph-bridge/callee-id-sink.js'; import { emitImportEdges } from '../graph-bridge/imports-to-edges.js'; import type { ScopeResolver } from '../contract/scope-resolver.js'; import { findEnclosingClassDef, resolveInheritanceBaseInScope } from '../scope/walkers.js'; @@ -411,6 +418,15 @@ interface RunScopeResolutionStats { * fixpoint phase composes them over the complete `CALLS` graph. */ readonly functionSummaries: readonly FunctionSummary[]; + /** + * Per-function RETURN-VALUE ASCENT summaries harvested in the pdg window + * (PDG FU-C, U-C2). Empty unless `input.pdg === true`. Keyed by resolved + * `Function`/`Method`/`Constructor` node id; the whole-program CALL_SUMMARY + * emit phase materialises one self-loop edge per entry once the call graph is + * known. Unlike {@link functionSummaries} this needs NO taint model — it is + * pure data-dependence — so it is harvested for every `--pdg` language. + */ + readonly callSummaries: readonly CallSummary[]; } export function runScopeResolution( @@ -529,6 +545,7 @@ export function runScopeResolution( referenceSkipped: 0, resolutionOutcomes, functionSummaries: [], + callSummaries: [], }; } @@ -701,6 +718,13 @@ export function runScopeResolution( // ── Phase 4: emit graph edges (LOAD-BEARING ORDER — see I1) ──────────── input.onProgress?.('linking symbols', files.length, files.length); const handledSites = new Set(preEmittedInheritanceSites); + // Resolved-callee-id capture accumulator (#2227 U2). Created ONLY under + // `--pdg` — `undefined` otherwise so the three emitters do zero work and emit + // byte-identical output (R4). Populated below at all three CALLS emit paths + // (each before its dedup, KTD6/R8); consumed by the CFG-emit join (U3) at + // `emitFileCfgs` below to produce `BasicBlock.calleeIds`. + const calleeIdAccumulator: CalleeIdAccumulator | undefined = + input.pdg === true ? createCalleeIdAccumulator() : undefined; const receiverExtras = emitReceiverBoundCalls( graph, indexes, @@ -712,6 +736,7 @@ export function runScopeResolution( readonlyModel, { recordResolutionOutcome, + calleeIdSink: calleeIdAccumulator, }, ); const unresolvedReceiverExtras = @@ -744,6 +769,7 @@ export function runScopeResolution( conversionOnlyArgTypePrefixes: provider.conversionOnlyArgTypePrefixes, constraintCompatibility: provider.constraintCompatibility, recordResolutionOutcome, + calleeIdSink: calleeIdAccumulator, }, ); const { emitted, skipped } = emitReferencesViaLookup( @@ -752,6 +778,7 @@ export function runScopeResolution( referenceIndex, postHeritageNodeLookup, handledSites, + calleeIdAccumulator, ); const importsEmitted = emitImportEdges( graph, @@ -788,6 +815,11 @@ export function runScopeResolution( // so the return (below the pdg block) can read it; empty on non-pdg runs. const harvestedSummaries: FunctionSummary[] = []; let summaryUnresolved = 0; + // FU-C (U-C2): per-function RETURN-VALUE ASCENT summaries harvested in the + // pdg window for the whole-program CALL_SUMMARY emit phase. Function-scoped + // (read by the return below the pdg block); empty on non-pdg runs. + const harvestedCallSummaries: CallSummary[] = []; + let callSummaryUnresolved = 0; // M3 (#2083 U4): accumulated taint time (match + taint-side solve + // propagate + TAINTED/SANITIZES emit), a sibling of `pdgMs` for the same // reason — it interleaves per file inside `emit=`, so only an accumulator @@ -854,10 +886,10 @@ export function runScopeResolution( // is built ONCE (whole-graph scan) and reused across every file; summaries // accumulate here and ride out on the stats for the cross-function fixpoint // phase. Only built when the language has a registered taint model. - const fnNodeIndex = - taintSpec !== undefined - ? (input.prebuiltFunctionNodeIndex ?? buildFunctionNodeIndex(graph)) - : undefined; + // Built whenever pdg is on (NOT gated on taintSpec): the FU-C call-summary + // harvest needs it for EVERY language (it is pure data-dependence, no taint + // model), and the taint summary harvest reuses it when taintSpec is present. + const fnNodeIndex = input.prebuiltFunctionNodeIndex ?? buildFunctionNodeIndex(graph); for (const pf of emitParsedFiles) { const cfgs = pf.cfgSideChannel; // Defensive: cfgSideChannel is opaque (`unknown`) and crosses the cache / @@ -892,6 +924,10 @@ export function runScopeResolution( ); } if (wellFormed.length === 0) continue; + // U3 hook (#2227): the resolved-callee-id map for this file is + // `calleeIdAccumulator?.get(pf.filePath)` — joined here by exact + // call-site position to emit `BasicBlock.calleeIds`. Captured above at + // the three CALLS emit paths (U2); wired into `emitFileCfgs` by U3. const emitted = emitFileCfgs( pdgTarget, wellFormed, @@ -900,16 +936,31 @@ export function runScopeResolution( // gated behind the semantic-model validator and silent in production) so // the per-function edge cap never truncates the CFG silently (R6/KTD6). (message) => logger.warn(message), + // U3 (#2227): the resolved-callee-id map for this file (captured at the + // three CALLS emit paths in U2), joined by exact call-site position to + // emit `BasicBlock.calleeIds`. `undefined` when pdg is off (the + // accumulator is only created under `input.pdg === true`). + calleeIdAccumulator?.get(pf.filePath), ); cfgBlocks += emitted.blocks; cfgEdges += emitted.edges; cfgDroppedEdges += emitted.droppedEdges; + // R6 (#2227 tri-review-2): release this file's captured id map now that + // emitFileCfgs has consumed it — the CALLS passes fully precede this loop + // and each file is read exactly once, so this bounds the accumulator to one + // file's call sites instead of holding the whole repo's for the phase. + calleeIdAccumulator?.delete(pf.filePath); // M2 (#2082 U4): reaching definitions over the same validated CFGs. // In-memory facts are computed per function and dropped after the // bounded (defBlock, useBlock, binding) projection is persisted — // M3 recomputes via the same pure solver in-phase (KTD8). Timing is // PROF-gated like every other checkpoint here (zero cost when off). + // U12: one memoized RD solver per file, shared by the RD-emit + call- + // summary + taint + summary passes, so the per-function fixpoint runs once + // per (limits) bucket instead of 3–4× (#2227 tri-review). File-scoped: it + // is re-created each iteration, so its per-function facts drop with the file. + const rdSolve = createMemoizedReachingDefs(); const t0 = PROF ? performance.now() : 0; const rd = emitFileReachingDefs( pdgTarget, @@ -917,6 +968,7 @@ export function runScopeResolution( input.pdgMaxReachingDefEdgesPerFunction ?? DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, (message) => logger.warn(message), // unconditional — R7, both layers + rdSolve, ); if (PROF) pdgMs += performance.now() - t0; rdEdges += rd.edges; @@ -941,6 +993,21 @@ export function runScopeResolution( cdgDropped += cdg.droppedEdges; cdgSkippedUnsound += cdg.skippedUnsoundFunctions; + // FU-C (U-C2): RETURN-VALUE ASCENT summaries over the SAME validated + // CFGs, inside the SAME per-file try. Independent of taint — runs for + // EVERY `--pdg` language (pure data-dependence, no source/sink model). + // Reuses the same RD fact cap the RD/taint solves use (coverage parity). + const callHarvest = harvestFileCallSummaries( + fnNodeIndex, + wellFormed, + taintLimits.maxFacts && taintLimits.maxFacts > 0 + ? taintLimits.maxFacts + : DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, + rdSolve, + ); + harvestedCallSummaries.push(...callHarvest.summaries); + callSummaryUnresolved += callHarvest.unresolved; + // M3 (#2083 U4): taint over the SAME validated CFGs, inside the SAME // per-file try (a taint throw costs this file's taint layer only — // its CFG/REACHING_DEF edges above are already in the graph). Skipped @@ -954,6 +1021,7 @@ export function runScopeResolution( taintSpec, taintLimits, (message) => logger.warn(message), // unconditional — R4/R6 + rdSolve, ); if (PROF) taintMs += performance.now() - t1; taintTotals.analyzed += taint.functionsAnalyzed; @@ -987,6 +1055,7 @@ export function runScopeResolution( taintLimits.maxFacts && taintLimits.maxFacts > 0 ? taintLimits.maxFacts : DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, + rdSolve, ); harvestedSummaries.push(...harvest.summaries); summaryUnresolved += harvest.unresolved; @@ -1090,6 +1159,16 @@ export function runScopeResolution( : ''), ); } + // FU-C (U-C2): call-summary harvest volume + anchor-resolution diagnostics. + if (harvestedCallSummaries.length > 0 || callSummaryUnresolved > 0) { + logger.debug( + `[call-summary] lang=${provider.language}: ${harvestedCallSummaries.length} function ` + + `return-ascent summary/summaries harvested` + + (callSummaryUnresolved > 0 + ? `, ${callSummaryUnresolved} CFG anchor(s) unresolved (same-line collision or missing node)` + : ''), + ); + } } if (PROF) { @@ -1120,5 +1199,6 @@ export function runScopeResolution( referenceSkipped: skipped, resolutionOutcomes, functionSummaries: harvestedSummaries, + callSummaries: harvestedCallSummaries, }; } diff --git a/gitnexus/src/core/ingestion/taint/call-summary-codec.ts b/gitnexus/src/core/ingestion/taint/call-summary-codec.ts new file mode 100644 index 000000000..4b2322574 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/call-summary-codec.ts @@ -0,0 +1,151 @@ +/** + * Call-summary reason codec (PDG FU-C, U-C2) — the ONE shared encoder/decoder + * for the bitset carried on persisted `CALL_SUMMARY` edges. + * + * A `CALL_SUMMARY` edge is a self-loop on a Function/Method/Constructor node + * recording that callee's RETURN-VALUE ASCENT: for each formal-parameter index, + * whether that parameter flows to the function's return value. The producer + * (the whole-program emit phase) writes this; a later consumer phase reads it to + * ascend a callee's return effect into the caller continuation. Two hand-rolled + * copies of a wire format drift — both sides MUST import from here (the same + * discipline `path-codec.ts` documents). + * + * ## Wire format (version `1`) + * + * ``` + * 1|r:[|…] + * ``` + * + * - One-character version prefix ({@link CALL_SUMMARY_CODEC_VERSION}), then `|`, + * then a `r:` (return) segment whose payload is the param→return bitset as a + * lowercase hex string (LSB = formal index 0). Bit `i` set ⇒ formal parameter + * `i` flows to the return value. An empty/zero bitset is the absence of any + * return-flow (a sound EMPTY summary — never a false claim). + * - FORWARD COMPATIBILITY: the format reserves space for future facts via + * additional trailing `|:` segments (planned: `o:` out-params, + * `e:` exception ascent — both deferred: out-params need an alias model, + * exception ascent needs try/catch CDG). The decoder accepts and ignores any + * trailing segment whose tag it does not understand, so a future writer's + * output stays decodable by today's reader (and vice-versa: today's reader + * only requires the `r:` segment). Reserved tags MUST stay disjoint from `r`. + * + * ## Delimiter / round-trip discipline (mirrors path-codec KTD6) + * + * Every structural character (`|`, `:`, the version digit, hex digits `[0-9a-f]`) + * is printable ASCII, so the encoding survives `escapeCSVField ∘ sanitizeUTF8` + * (csv-generator.ts) byte-exact (pinned by the round-trip test). The hex payload + * is a non-negative integer rendered via BigInt, so the codec handles functions + * with arbitrarily many formal parameters without overflow. + * + * The decoder NEVER throws — anything malformed yields a typed failure, exactly + * like `decodeTaintPath`. A decode failure on the consumer side means "no usable + * ascent fact", which is the sound default (never claim a false return-flow). + */ + +/** One-character format version prefix. Bump on any wire-format change. */ +export const CALL_SUMMARY_CODEC_VERSION = '1'; + +/** Return-flow segment tag (`r:`). */ +const RETURN_TAG = 'r'; + +/** Hex payload charset (lowercase). */ +const HEX = /^[0-9a-f]+$/; + +/** A decoded call summary's facts. Forward-compatible: future facts add fields. */ +export interface DecodedCallSummary { + readonly ok: true; + readonly version: string; + /** + * Sorted, de-duplicated formal-parameter indices that flow to the return + * value (ascending). Empty ⇒ no return-flow recorded (sound EMPTY summary). + */ + readonly returnFlowParams: readonly number[]; +} + +/** Typed parse failure — the decoder never throws. */ +export interface CallSummaryDecodeFailure { + readonly ok: false; + readonly error: string; +} + +export type CallSummaryDecodeResult = DecodedCallSummary | CallSummaryDecodeFailure; + +/** + * Pack a set of return-flowing formal-parameter indices into a bitset (LSB = + * index 0). Negative or non-integer indices are ignored (defensive; the + * harvester only ever passes non-negative integers). Returns a `BigInt`. + */ +function packReturnBitset(returnFlowParams: Iterable): bigint { + let bits = 0n; + for (const idx of returnFlowParams) { + if (!Number.isInteger(idx) || idx < 0) continue; + bits |= 1n << BigInt(idx); + } + return bits; +} + +/** + * Encode the param→return ascent into the versioned `reason` wire string. + * Deterministic; never throws. `returnFlowParams` is the set of formal indices + * that flow to the return value (order/duplication irrelevant — the bitset + * canonicalises). An empty set encodes as `r:0` (an explicit empty summary). + */ +export function encodeCallSummary(returnFlowParams: Iterable): string { + const bits = packReturnBitset(returnFlowParams); + return `${CALL_SUMMARY_CODEC_VERSION}|${RETURN_TAG}:${bits.toString(16)}`; +} + +/** + * Decode a `CALL_SUMMARY` reason wire string into its ascent facts. Returns a + * typed failure for anything that is not a well-formed version-`1` summary — + * never throws. Unknown trailing segments (future facts) are accepted and + * ignored (forward compatibility). + */ +export function decodeCallSummary(reason: unknown): CallSummaryDecodeResult { + if (typeof reason !== 'string' || reason.length === 0) { + return { ok: false, error: 'empty or non-string reason' }; + } + // Read the version as the substring BEFORE the first '|' (NOT a single char): + // the wire format reserves multi-digit future versions, so a `12|…` writer must + // degrade to a clean 'unsupported version' typed-failure here, never silently + // parse as version '1' with a stray '2' segment. No '|' ⇒ the whole reason is a + // bare token with no body, which is malformed. + const firstSep = reason.indexOf('|'); + if (firstSep === -1) { + return { ok: false, error: 'malformed body: expected a segment separator after the version' }; + } + const version = reason.slice(0, firstSep); + if (version !== CALL_SUMMARY_CODEC_VERSION) { + return { ok: false, error: `unsupported call-summary version '${version}'` }; + } + const segments = reason.slice(firstSep + 1).split('|'); + let returnBits: bigint | undefined; + for (const seg of segments) { + const sep = seg.indexOf(':'); + if (sep === -1) { + return { ok: false, error: `malformed segment '${seg}' (expected ':')` }; + } + const tag = seg.slice(0, sep); + const payload = seg.slice(sep + 1); + if (tag === RETURN_TAG) { + if (!HEX.test(payload)) { + return { ok: false, error: `invalid return bitset '${payload}'` }; + } + returnBits = BigInt(`0x${payload}`); + } + // Unknown tag (reserved future fact): accept + ignore for forward compat. + } + if (returnBits === undefined) { + return { ok: false, error: "missing required 'r:' return segment" }; + } + // Unpack the bitset into ascending formal indices. + const returnFlowParams: number[] = []; + let bits = returnBits; + let idx = 0; + while (bits > 0n) { + if ((bits & 1n) === 1n) returnFlowParams.push(idx); + bits >>= 1n; + idx++; + } + return { ok: true, version, returnFlowParams }; +} diff --git a/gitnexus/src/core/ingestion/taint/call-summary-emit.ts b/gitnexus/src/core/ingestion/taint/call-summary-emit.ts new file mode 100644 index 000000000..4c8d4e914 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/call-summary-emit.ts @@ -0,0 +1,99 @@ +/** + * Call-summary emission (PDG FU-C, U-C3) — materialise `CALL_SUMMARY`. + * + * Persists each per-callee {@link CallSummary} as ONE `CALL_SUMMARY` self-loop + * edge on the callee Function/Method/Constructor node, with the param→return + * ASCENT bitset encoded in `reason` via the SHARED {@link call-summary-codec} + * (never a second hand-rolled wire format). A later consumer phase decodes it to + * ascend a callee's return effect into the caller continuation. + * + * The self-loop shape matches the schema (Function→Function / Method→Method / + * Constructor→Constructor pairs already exist from the M0 TAINT_PATH work — zero + * new schema pairs). Like `TAINT_PATH`/`TAINTED`, `CALL_SUMMARY` stays OUT of + * `VALID_RELATION_TYPES` and the web schema — it is an internal PDG-engine edge. + * + * Boundedness mirrors the M4 interproc emit driver: dedup by the deterministic + * edge id, an optional per-run cap, and unconditional truncate-and-warn. + */ + +import type { KnowledgeGraph } from '../../graph/types.js'; +import { encodeCallSummary } from './call-summary-codec.js'; +import type { CallSummary } from './call-summary-model.js'; + +/** Confidence stamped on `CALL_SUMMARY` edges. A summary is a context- + * insensitive whole-parameter abstraction — a coarser signal than a resolved + * `CALLS` edge, so kept below 1.0 (mirrors the interproc TAINT_PATH posture). */ +export const CALL_SUMMARY_CONFIDENCE = 0.6; + +/** Default per-run cap on emitted `CALL_SUMMARY` edges. `0` ⇒ unlimited. */ +export const DEFAULT_PDG_MAX_CALL_SUMMARY_EDGES = 0; + +export interface CallSummaryEmitLimits { + /** Max `CALL_SUMMARY` edges per run (post-dedup). `undefined`/0 ⇒ unlimited. */ + readonly maxEdges?: number; +} + +export interface CallSummaryEmitResult { + /** CALL_SUMMARY edges persisted. */ + edgesEmitted: number; + /** Summaries dropped by the per-run cap. */ + edgesDropped: number; + /** Summaries skipped because the callee node was missing from the graph. */ + skippedMissingEndpoint: number; +} + +/** + * Persist per-callee summaries as `CALL_SUMMARY` self-loop edges. `summaries` is + * assumed deterministically ordered (the harvest sorts `returnFlowParams`). + * Never throws on valid input. + */ +export function emitCallSummaries( + graph: KnowledgeGraph, + summaries: readonly CallSummary[], + limits?: CallSummaryEmitLimits, + onWarn?: (message: string) => void, +): CallSummaryEmitResult { + const result: CallSummaryEmitResult = { + edgesEmitted: 0, + edgesDropped: 0, + skippedMissingEndpoint: 0, + }; + const maxEdges = limits?.maxEdges && limits.maxEdges > 0 ? limits.maxEdges : Infinity; + const seen = new Set(); + + for (const summary of summaries) { + if (result.edgesEmitted >= maxEdges) { + result.edgesDropped++; + continue; + } + const node = graph.getNode(summary.fnId); + if (!node) { + result.skippedMissingEndpoint++; + continue; + } + // One self-loop edge per callee; dedup by the callee id (the harvest already + // produces at most one summary per resolved fnId, but a same-line anchor + // could in principle map two CFGs to one id — the Set keeps it idempotent). + const id = `rel:CALL_SUMMARY:${summary.fnId}`; + if (seen.has(id)) continue; + seen.add(id); + + graph.addRelationship({ + id, + sourceId: summary.fnId, + targetId: summary.fnId, + type: 'CALL_SUMMARY', + confidence: CALL_SUMMARY_CONFIDENCE, + reason: encodeCallSummary(summary.returnFlowParams), + }); + result.edgesEmitted++; + } + + if (result.edgesDropped > 0) { + onWarn?.( + `[call-summary] ${result.edgesDropped} CALL_SUMMARY edge(s) dropped by the ` + + `per-run cap (${maxEdges})`, + ); + } + return result; +} diff --git a/gitnexus/src/core/ingestion/taint/call-summary-harvest.ts b/gitnexus/src/core/ingestion/taint/call-summary-harvest.ts new file mode 100644 index 000000000..6bef98dce --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/call-summary-harvest.ts @@ -0,0 +1,210 @@ +/** + * Per-function dependence-SUMMARY harvest (PDG FU-C, U-C2). + * + * Pure, deterministic derivation of one function's RETURN-VALUE ASCENT — which + * formal-parameter indices flow to the function's return value — from the SAME + * substrate the M2/M3 passes consume: the reaching-definition facts + * (`computeReachingDefs`) over the function's CFG. No graph, no I/O, no logger; + * mirrors the {@link harvestFunctionSummary} (taint) contract so snapshot tests + * and the version stamp stay stable. Runs IN-PHASE inside the scope-resolution + * pdg window where the RD facts are materialised (reusing them — zero new + * worker/CFG work, so NO parse-cache pdg:N bump). + * + * ## Return-site identification (language-agnostic, soundness-first) + * + * Return statements are identified STRUCTURALLY via the M2 edge-kind invariant: + * the SOURCE block of every CFG edge of kind `return` terminates in the return + * jump, so that block's LAST statement is the `return ` — its `uses` are + * the returned bindings. A `return;` with no value has empty uses (contributes + * nothing). For languages whose visitor models IMPLICIT returns (arrow-function + * expression bodies, Python last-expression), the CFG emits a `return` edge to + * EXIT whose source block's last statement carries the returned expression's + * `uses`, so those flow through the same path with no language-specific code. + * + * SOUNDNESS = never claim a false return-flow: when a function has NO `return` + * CFG edge (a language/shape with no robust exit notion modelled, or a void + * function), `returnUseStmtKeys` is empty and the harvest emits an EMPTY summary + * — the absence of a fact, never a wrong one. + * + * ## Param → return reachability + * + * Each formal parameter is seeded as a value at its entry def point(s); forward + * reachability over the def→use facts marks the param's index as return-flowing + * the moment a tainted binding it produced (under the M3 statement-level floor: + * a statement using a value taints all of its defs/mayDefs) is among a + * return-use statement's `uses`. The recorded edge is from an ACTUAL binding + * occurrence in a return's uses — never the floor — keeping the recorded fact + * precise even though onward propagation over-approximates. + * + * ## Formal-position soundness — destructured / rest params + * + * The consumer reads `returnFlowParams` POSITIONALLY (call-site arg position → + * same-index formal → bitset), so each recorded index MUST be the 0-based + * ENCLOSING FORMAL position, never the flattened binding ordinal. A + * destructured/rest formal binds several names: `function f({a, b}, c)` flattens + * to bindings a, b, c, whose ORDINALS are 0, 1, 2 — but the formal positions are + * 0, 0, 1. Recording an ordinal would misattribute `b`'s return-flow to formal + * `c` (a FALSE return-flow claim, not a miss). To stay sound we key every + * recorded index on {@link BindingEntry.formalIndex} (the producer-supplied + * enclosing-formal position, identical for every inner name of one formal). + * + * CONSERVATIVE FALLBACK: a producer that does not yet supply `formalIndex` on + * its param bindings leaves the harvest unable to prove the ordinal equals the + * formal slot, so the harvest emits an EMPTY summary for that function — a + * documented MISS (loses ascent), NEVER a false claim. Functions whose every + * param binding carries `formalIndex` get the precise formal positions. + */ + +import type { FunctionCfg } from '../cfg/types.js'; +import { pointKey, type FunctionDefUse, type ProgramPoint } from '../cfg/reaching-defs.js'; + +/** The own-facts portion of a call summary (fnId/anchor added by the caller). */ +export interface HarvestedCallSummaryFacts { + readonly paramCount: number; + /** Sorted, de-duplicated formal-parameter indices that flow to the return. */ + readonly returnFlowParams: readonly number[]; +} + +export interface CallSummaryHarvestResult { + /** `computed` — facts derived; `coverage-gap` — the RD solver was not + * `computed`, so no summary is produced (consistent with the taint harvest). */ + readonly status: 'computed' | 'coverage-gap'; + readonly gapReason?: FunctionDefUse['status']; + readonly facts: HarvestedCallSummaryFacts; +} + +const EMPTY_FACTS: HarvestedCallSummaryFacts = { paramCount: 0, returnFlowParams: [] }; + +/** A value flowing forward, tagged with the param seed it came from. */ +interface SeedValue { + readonly bindingIdx: number; + readonly point: ProgramPoint; + /** Param index (≥0) this value originates from. */ + readonly paramIdx: number; +} + +/** + * Harvest the RETURN-VALUE ASCENT facts for one function. PRECONDITION: `cfg` + * is `isEmitSafeCfg`-filtered and `defUse` was computed from it (the caller + * gates exactly as the taint harvest path does). + */ +export function harvestCallSummary( + cfg: FunctionCfg, + defUse: FunctionDefUse, +): CallSummaryHarvestResult { + if (defUse.status !== 'computed') { + return { status: 'coverage-gap', gapReason: defUse.status, facts: EMPTY_FACTS }; + } + const bindings = defUse.bindings; + + // ── param bindings → ENCLOSING FORMAL position ──────────────────────────── + // SOUNDNESS (FU-C): the consumer joins `returnFlowParams` positionally against + // call-site arg positions, so each index MUST be the 0-based enclosing formal + // position — `BindingEntry.formalIndex`, which a destructured/rest formal hands + // identically to every inner name. The flattened binding ORDINAL is NOT a safe + // substitute (`function f({a, b}, c)` ⇒ b's ordinal 1 collides with formal c). + const paramBindings = bindings + .map((b, idx) => ({ b, idx })) + .filter((e) => e.b.kind === 'param') + .sort((a, b) => a.b.declLine - b.b.declLine || a.b.declColumn - b.b.declColumn); + // CONSERVATIVE FALLBACK: build binding-index → enclosing-formal-position only + // while every param binding supplies `formalIndex` (narrowed per-entry, no + // assertion). If ANY lacks it, the ordinal-vs-formal mapping is unprovable, so + // the summary degrades to EMPTY below (a documented MISS, never a false claim). + const paramCount = paramBindings.length; + const paramFormalOf = new Map(); + let missingFormalIndex = false; + for (const e of paramBindings) { + const formalIndex = e.b.formalIndex; + if (formalIndex === undefined) { + missingFormalIndex = true; + break; + } + paramFormalOf.set(e.idx, formalIndex); + } + + // ── return points: source block of every `return` CFG edge ──────────────── + // The M2 edge-kind invariant: a `return` edge's SOURCE block terminates in the + // return jump, so its LAST statement is `return ` — its `uses` are the + // returned bindings. (`return;` with no value has empty uses.) No `return` + // edge ⇒ empty set ⇒ EMPTY summary (sound — never a false return-flow claim). + const returnUseStmtKeys = new Set(); + for (const e of cfg.edges) { + if (e.kind !== 'return') continue; + const block = cfg.blocks[e.from]; + const stmts = block?.statements; + if (!stmts || stmts.length === 0) continue; + returnUseStmtKeys.add(`${e.from}:${stmts.length - 1}`); + } + + // Fast exit: no params, no return sites, or an unprovable formal mapping + // (conservative fallback) ⇒ nothing safely flows to the return. + if (paramCount === 0 || returnUseStmtKeys.size === 0 || missingFormalIndex) { + return { status: 'computed', facts: { paramCount, returnFlowParams: [] } }; + } + + const stmtAt = (p: ProgramPoint) => cfg.blocks[p.blockIndex]?.statements?.[p.stmtIndex]; + + // ── def→use index ───────────────────────────────────────────────────────── + const factsByDef = new Map(); + for (const f of defUse.facts) { + const key = `${f.bindingIdx}:${pointKey(f.def)}`; + const list = factsByDef.get(key); + const entry = { bindingIdx: f.bindingIdx, use: f.use }; + if (list) list.push(entry); + else factsByDef.set(key, [entry]); + } + + // ── seeds: each param at its entry def point(s) ──────────────────────────── + const queue: SeedValue[] = []; + const visited = new Set(); + const enqueue = (v: SeedValue): void => { + const key = `${v.paramIdx}:${v.bindingIdx}:${pointKey(v.point)}`; + if (visited.has(key)) return; + visited.add(key); + queue.push(v); + }; + for (const { idx } of paramBindings) { + // 0-based ENCLOSING formal position — guaranteed present (the missing-formal + // case returned EMPTY above), but guard rather than assert to stay `any`-free. + const paramIdx = paramFormalOf.get(idx); + if (paramIdx === undefined) continue; + for (const f of defUse.facts) { + if (f.bindingIdx === idx && f.def.blockIndex === cfg.entryIndex) { + enqueue({ bindingIdx: idx, point: f.def, paramIdx }); + } + } + } + + // ── forward reachability ────────────────────────────────────────────────── + const returnFlow = new Set(); + let head = 0; + while (head < queue.length) { + const v = queue[head++]; + const b = v.bindingIdx; + for (const fact of factsByDef.get(`${b}:${pointKey(v.point)}`) ?? []) { + const useStmt = stmtAt(fact.use); + if (!useStmt) continue; + const useKey = `${fact.use.blockIndex}:${fact.use.stmtIndex}`; + // (1) return reach: the param's value is among a return-use's `uses`. + if (returnUseStmtKeys.has(useKey) && useStmt.uses.includes(b)) { + returnFlow.add(v.paramIdx); + } + // (2) onward floor: this statement's defs/mayDefs carry the value onward. + for (const d of [...useStmt.defs, ...(useStmt.mayDefs ?? [])]) { + enqueue({ + bindingIdx: d, + point: { + blockIndex: fact.use.blockIndex, + stmtIndex: fact.use.stmtIndex, + line: useStmt.line, + }, + paramIdx: v.paramIdx, + }); + } + } + } + + const returnFlowParams = [...returnFlow].sort((a, b) => a - b); + return { status: 'computed', facts: { paramCount, returnFlowParams } }; +} diff --git a/gitnexus/src/core/ingestion/taint/call-summary-model.ts b/gitnexus/src/core/ingestion/taint/call-summary-model.ts new file mode 100644 index 000000000..5d498cc25 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/call-summary-model.ts @@ -0,0 +1,49 @@ +/** + * Per-callee dependence SUMMARY model (PDG FU-C, U-C2). + * + * A {@link CallSummary} is the compact, context-insensitive abstraction of one + * function's RETURN-VALUE ASCENT: which formal-parameter indices flow to the + * function's return value. It is the data-dependence twin of the M4 + * {@link FunctionSummary} (taint), but for the *slicing* engine rather than the + * taint engine — a later consumer phase uses it to ascend a callee's return + * effect into the caller continuation (the documented no-ascent false negative). + * + * ## Scope (first cut — RETURN-VALUE ONLY) + * + * WHOLE-PARAMETER granularity. Ports are `param i` → `return`. Out-params / + * mutated args (need an alias model) and exception ascent (need try/catch CDG) + * are DEFERRED — the {@link call-summary-codec} reserves wire-format space for + * them so they can land without a cache-namespace bump. + * + * ## Plain-data discipline + * + * A summary is a JSON-plain value type (no functions, class instances, Maps, or + * Symbols) so it survives `RunScopeResolutionStats` → `ScopeResolutionOutput` + * threading unchanged — the same `Cloneable` constraint the CFG side channel and + * the taint {@link FunctionSummary} obey. + */ + +/** Source-relative parameter index (0-based, declaration order). */ +export type ParamIndex = number; + +/** + * The dependence abstraction of one function. The resolved Function/Method/ + * Constructor graph node id this summary describes, plus the set of formal + * parameters whose value flows to the return. + */ +export interface CallSummary { + /** The resolved `Function`/`Method`/`Constructor` graph node id. */ + readonly fnId: string; + /** Repo-relative source path (carried for diagnostics + the anchor join). */ + readonly filePath: string; + /** 1-based function start line (mirrors `FunctionCfg.functionStartLine`). */ + readonly startLine: number; + /** Number of declared formal parameters (port arity). */ + readonly paramCount: number; + /** + * Sorted, de-duplicated formal-parameter indices that flow to the function's + * return value (ascending). Empty ⇒ no parameter reaches the return (a sound + * EMPTY summary — never a false claim). + */ + readonly returnFlowParams: readonly ParamIndex[]; +} diff --git a/gitnexus/src/core/ingestion/taint/emit.ts b/gitnexus/src/core/ingestion/taint/emit.ts index d11694b28..a1981f35f 100644 --- a/gitnexus/src/core/ingestion/taint/emit.ts +++ b/gitnexus/src/core/ingestion/taint/emit.ts @@ -71,7 +71,12 @@ import { bindingKey, DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, } from '../cfg/emit.js'; -import { computeReachingDefs, pointKey, type ProgramPoint } from '../cfg/reaching-defs.js'; +import { + computeReachingDefs, + pointKey, + type ProgramPoint, + type ReachingDefsSolver, +} from '../cfg/reaching-defs.js'; import type { BindingEntry, FunctionCfg } from '../cfg/types.js'; import { hasTaintSafeSites } from './site-safety.js'; import { buildTaintImportIndex, matchFunctionSites } from './match.js'; @@ -156,6 +161,10 @@ export function emitFileTaint( spec: SourceSinkSanitizerSpec, limits?: TaintEmitLimits, onWarn?: (message: string) => void, + // U12: shared per-file memoized solver (harvest/taint bucket — no maxBlockVisits). + // The zero-match fast path below still skips the solve entirely; only MATCHED + // functions request it, hitting the cache the call-summary harvest warmed. + solve: ReachingDefsSolver = computeReachingDefs, ): TaintEmitResult { const result: TaintEmitResult = { functionsAnalyzed: 0, @@ -204,7 +213,7 @@ export function emitFileTaint( continue; } - const defUse = computeReachingDefs(cfg, { maxFacts }); + const defUse = solve(cfg, { maxFacts }); const flows = computeTaintFlows(cfg, defUse, matches, { maxFindingsPerFunction, maxHops }); if (flows.status === 'coverage-gap') { // R4: skipped entirely, counted by reason; aggregate-warned by the diff --git a/gitnexus/src/core/ingestion/taint/summary-harvest-driver.ts b/gitnexus/src/core/ingestion/taint/summary-harvest-driver.ts index 8b5a3f864..5ded4bd15 100644 --- a/gitnexus/src/core/ingestion/taint/summary-harvest-driver.ts +++ b/gitnexus/src/core/ingestion/taint/summary-harvest-driver.ts @@ -30,18 +30,29 @@ import type { ParsedImport, GraphNode } from 'gitnexus-shared'; import type { KnowledgeGraph } from '../../graph/types.js'; -import { computeReachingDefs } from '../cfg/reaching-defs.js'; +import { computeReachingDefs, type ReachingDefsSolver } from '../cfg/reaching-defs.js'; import { DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION } from '../cfg/emit.js'; import type { FunctionCfg } from '../cfg/types.js'; import { buildTaintImportIndex, matchFunctionSites } from './match.js'; import type { SourceSinkSanitizerSpec } from './source-sink-config.js'; import { harvestFunctionSummary } from './summary-harvest.js'; import { ownFactsDigest, summaryVersion, type FunctionSummary } from './summary-model.js'; +import { harvestCallSummary } from './call-summary-harvest.js'; +import type { CallSummary } from './call-summary-model.js'; /** `cfg.functionStartLine` (1-based) − this = the node's 0-based `startLine`. */ export const NODE_TO_CFG_LINE_OFFSET = 1; -/** Node labels that can own a CFG / be a `CALLS` endpoint. */ +/** + * Node labels that can own a CFG / be a `CALLS` endpoint AND receive a + * return-value-ascent summary. `Constructor` is INTENTIONALLY excluded: a + * constructor's "return" is the freshly-allocated instance, not a user-flowed + * value, so a formal→return ascent is not meaningful for it. A Constructor CFG + * therefore resolves to no functionish node (counted `unresolved`) and emits no + * CALL_SUMMARY edge — a sound recall miss, never a false ascent. The impact + * consumer may still DESCEND into a Constructor; it just never learns a + * constructor's return-flow. (#2227 tri-review.) + */ const FUNCTIONISH_LABELS = new Set(['Function', 'Method']); /** @@ -99,6 +110,8 @@ export function harvestFileSummaries( parsedImports: readonly ParsedImport[], spec: SourceSinkSanitizerSpec, maxFacts: number = DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, + // U12: shared per-file memoized solver (harvest/taint bucket — no maxBlockVisits). + solve: ReachingDefsSolver = computeReachingDefs, ): FileSummaryResult { const importIndex = buildTaintImportIndex(parsedImports); const summaries: FunctionSummary[] = []; @@ -111,7 +124,7 @@ export function harvestFileSummaries( unresolved++; continue; } - const defUse = computeReachingDefs(cfg, { maxFacts }); + const defUse = solve(cfg, { maxFacts }); const matches = matchFunctionSites(cfg, spec, importIndex); const harvested = harvestFunctionSummary(cfg, defUse, matches); if (harvested.status !== 'computed') { @@ -145,3 +158,66 @@ export function harvestFileSummaries( return { summaries, unresolved, gaps }; } + +export interface FileCallSummaryResult { + readonly summaries: readonly CallSummary[]; + /** CFGs whose anchor resolved to no unique graph node (collision / missing). */ + readonly unresolved: number; + /** CFGs whose reaching-defs were not `computed` (no summary produced). */ + readonly gaps: number; +} + +/** + * Harvest per-function RETURN-VALUE ASCENT summaries (PDG FU-C, U-C2) for one + * file's emit-safe CFGs — the dependence-engine SIBLING of + * {@link harvestFileSummaries}. `cfgs` MUST already be `isEmitSafeCfg`-filtered. + * Pure aside from the read-only graph lookup; never throws on valid input. + * + * Unlike the taint harvest, this needs NO source/sink model — return-value + * ascent is purely data-dependence over the RD facts — so it runs for every + * `--pdg` language (not just those with a registered taint spec). It reuses the + * SAME per-function RD facts (recomputed via the same pure solver + cap the RD + * emit used; the persisted REACHING_DEF projection is a lossy subset, so the + * harvest re-derives in-phase exactly as the taint harvest does — no new + * worker/CFG work, no parse-cache bump). + */ +export function harvestFileCallSummaries( + fnIndex: FunctionNodeIndex, + cfgs: readonly FunctionCfg[], + maxFacts: number = DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, + // U12: shared per-file memoized solver (harvest/taint bucket — no maxBlockVisits). + solve: ReachingDefsSolver = computeReachingDefs, +): FileCallSummaryResult { + const summaries: CallSummary[] = []; + let unresolved = 0; + let gaps = 0; + + for (const cfg of cfgs) { + const fnId = resolveFnId(fnIndex, cfg); + if (fnId === undefined) { + unresolved++; + continue; + } + const defUse = solve(cfg, { maxFacts }); + const harvested = harvestCallSummary(cfg, defUse); + if (harvested.status !== 'computed') { + gaps++; + continue; + } + const facts = harvested.facts; + // Skip functions with NO return-flow at all — an empty summary records no + // ascent fact, so persisting it would only bloat the edge set. (The consumer + // treats an absent CALL_SUMMARY as "no known ascent", identical to an empty + // one.) A 0-param function or a void function therefore emits no edge. + if (facts.returnFlowParams.length === 0) continue; + summaries.push({ + fnId, + filePath: cfg.filePath, + startLine: cfg.functionStartLine, + paramCount: facts.paramCount, + returnFlowParams: facts.returnFlowParams, + }); + } + + return { summaries, unresolved, gaps }; +} diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index 0dc6eba94..a03a81ada 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -261,11 +261,16 @@ export const buildRelRow = (rel: GraphRelationship): string => * No `name` column; blocks are identified by id + source span. Shared by the * whole-graph emit pass and the streaming PDG emit sink (issue #2202) so the * two paths produce byte-identical BasicBlock rows by construction. */ -export const BASICBLOCK_CSV_HEADER = 'id,filePath,startLine,endLine,text'; +export const BASICBLOCK_CSV_HEADER = 'id,filePath,startLine,endLine,text,callees,calleeIds'; /** Build the escaped CSV row (no trailing newline) for one BasicBlock node. * Single source of the BasicBlock row bytes — used by `streamAllCSVsToDisk` - * and by the streaming `PdgEmitSink` (issue #2202). */ + * and by the streaming `PdgEmitSink` (issue #2202). `callees` is a comma-free + * (space-joined) list of the leaf callee names invoked in the block — the + * statement-precise inter-procedural reach substrate (the field is itself a CSV + * cell, so the inner separator must NOT be a comma). `calleeIds` is the SOUND + * parallel to `callees`: the space-joined RESOLVED callee symbol ids for the + * block (#2227 follow-up), likewise a comma-free cell. */ export const buildBasicBlockRow = (node: GraphNode): string => [ escapeCSVField(node.id), @@ -273,6 +278,8 @@ export const buildBasicBlockRow = (node: GraphNode): string => escapeCSVNumber(node.properties.startLine, -1), escapeCSVNumber(node.properties.endLine, -1), escapeCSVField(node.properties.text || ''), + escapeCSVField(String(node.properties.callees ?? '')), + escapeCSVField(String(node.properties.calleeIds ?? '')), ].join(','); export interface StreamedCSVResult { @@ -374,7 +381,7 @@ export const streamAllCSVsToDisk = async ( // Route nodes for API endpoint mapping const routeWriter = new BufferedCSVWriter( path.join(csvDir, 'route.csv'), - 'id,name,filePath,responseKeys,errorKeys,middleware', + 'id,name,filePath,responseKeys,errorKeys,middleware,method', ); // Tool nodes for MCP tool definitions @@ -553,6 +560,7 @@ export const streamAllCSVsToDisk = async ( escapeCSVField(keysStr), escapeCSVField(errorKeysStr), escapeCSVField(middlewareStr), + escapeCSVField(String(node.properties.method ?? '')), ].join(','), ); break; diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index f5ff27281..eb04ec5a4 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -1305,14 +1305,16 @@ export const getCopyQuery = (table: NodeTableName, filePath: string): string => return `COPY ${t}(id, name, filePath, startLine, endLine, level, content, description) FROM "${filePath}" ${COPY_CSV_OPTS}`; } if (table === 'Route') { - return `COPY ${t}(id, name, filePath, responseKeys, errorKeys, middleware) FROM "${filePath}" ${COPY_CSV_OPTS}`; + return `COPY ${t}(id, name, filePath, responseKeys, errorKeys, middleware, method) FROM "${filePath}" ${COPY_CSV_OPTS}`; } if (table === 'Tool') { return `COPY ${t}(id, name, filePath, description) FROM "${filePath}" ${COPY_CSV_OPTS}`; } if (table === 'BasicBlock') { - // Taint/PDG substrate (issue #2080) — no name column. - return `COPY ${t}(id, filePath, startLine, endLine, text) FROM "${filePath}" ${COPY_CSV_OPTS}`; + // Taint/PDG substrate (issue #2080) — no name column. `callees` is the + // statement-precise inter-procedural reach substrate (space-joined leaf names); + // `calleeIds` is its SOUND parallel (space-joined resolved callee ids, #2227). + return `COPY ${t}(id, filePath, startLine, endLine, text, callees, calleeIds) FROM "${filePath}" ${COPY_CSV_OPTS}`; } if (table === 'Method') { return `COPY ${t}(id, name, filePath, startLine, endLine, isExported, content, description, parameterCount, returnType) FROM "${filePath}" ${COPY_CSV_OPTS}`; @@ -1367,8 +1369,9 @@ export const insertNodeToLbug = async ( : ''; query = `CREATE (n:Section {id: ${escapeValue(properties.id)}, name: ${escapeValue(properties.name)}, filePath: ${escapeValue(properties.filePath)}, startLine: ${properties.startLine || 0}, endLine: ${properties.endLine || 0}, level: ${properties.level || 1}, content: ${escapeValue(properties.content || '')}${descPart}})`; } else if (label === 'BasicBlock') { - // Taint/PDG substrate (issue #2080) — no name column. - query = `CREATE (n:BasicBlock {id: ${escapeValue(properties.id)}, filePath: ${escapeValue(properties.filePath)}, startLine: ${properties.startLine || 0}, endLine: ${properties.endLine || 0}, text: ${escapeValue(properties.text || '')}})`; + // Taint/PDG substrate (issue #2080) — no name column. `calleeIds` (#2227) + // is the sound resolved-id parallel to the leaf-name `callees` set. + query = `CREATE (n:BasicBlock {id: ${escapeValue(properties.id)}, filePath: ${escapeValue(properties.filePath)}, startLine: ${properties.startLine || 0}, endLine: ${properties.endLine || 0}, text: ${escapeValue(properties.text || '')}, callees: ${escapeValue(properties.callees || '')}, calleeIds: ${escapeValue(properties.calleeIds || '')}})`; } else if (TABLES_WITH_EXPORTED.has(label)) { const descPart = properties.description ? `, description: ${escapeValue(properties.description)}` @@ -1453,8 +1456,9 @@ export const batchInsertNodesToLbug = async ( : ''; query = `MERGE (n:Section {id: ${escapeValue(properties.id)}}) SET n.name = ${escapeValue(properties.name)}, n.filePath = ${escapeValue(properties.filePath)}, n.startLine = ${properties.startLine || 0}, n.endLine = ${properties.endLine || 0}, n.level = ${properties.level || 1}, n.content = ${escapeValue(properties.content || '')}${descPart}`; } else if (label === 'BasicBlock') { - // Taint/PDG substrate (issue #2080) — no name column. - query = `MERGE (n:BasicBlock {id: ${escapeValue(properties.id)}}) SET n.filePath = ${escapeValue(properties.filePath)}, n.startLine = ${properties.startLine || 0}, n.endLine = ${properties.endLine || 0}, n.text = ${escapeValue(properties.text || '')}`; + // Taint/PDG substrate (issue #2080) — no name column. `calleeIds` + // (#2227) is the sound resolved-id parallel to the `callees` set. + query = `MERGE (n:BasicBlock {id: ${escapeValue(properties.id)}}) SET n.filePath = ${escapeValue(properties.filePath)}, n.startLine = ${properties.startLine || 0}, n.endLine = ${properties.endLine || 0}, n.text = ${escapeValue(properties.text || '')}, n.callees = ${escapeValue(properties.callees || '')}, n.calleeIds = ${escapeValue(properties.calleeIds || '')}`; } else if (TABLES_WITH_EXPORTED.has(label)) { const descPart = properties.description ? `, n.description = ${escapeValue(properties.description)}` @@ -2072,6 +2076,56 @@ export const deleteAllInterprocTaintPaths = async (): Promise<{ edgesDeleted: nu return { edgesDeleted }; }; +/** + * Drop every `CALL_SUMMARY` relationship (PDG FU-C, U-C3). Used at the start of + * an incremental `--pdg` writeback so the `callSummaries` phase re-materialises + * them from scratch on the FULL recomputed graph. + * + * Mirrors {@link deleteAllInterprocTaintPaths}: CALL_SUMMARY is a self-loop edge + * type (not a node label), so a plain DELETE on the typed CodeRelation rows + * leaves endpoints untouched. `extractChangedSubgraph` re-includes ALL of them + * from the fresh graph (`isGraphWideRelType`), so delete-all-then-rebuild keeps + * an unchanged function's summary from being lost. + */ +export const deleteAllCallSummaries = async (): Promise<{ edgesDeleted: number }> => { + if (!conn) { + throw new Error('LadybugDB not initialized. Call initLbug first.'); + } + let edgesDeleted = 0; + let countResult: lbug.QueryResult | lbug.QueryResult[] | undefined; + try { + countResult = await conn.query( + `MATCH ()-[r:CodeRelation]->() WHERE r.type = 'CALL_SUMMARY' RETURN count(r) AS cnt`, + ); + const result = Array.isArray(countResult) ? countResult[0] : countResult; + const rows = await result.getAll(); + const count = Number(rows[0]?.cnt ?? rows[0]?.[0] ?? 0); + if (count > 0) { + await conn.query(`MATCH ()-[r:CodeRelation]->() WHERE r.type = 'CALL_SUMMARY' DELETE r`); + edgesDeleted = count; + } + } catch (err) { + // A missing table on a freshly-initialized DB is the benign, expected case + // (the count query is what throws) — stay silent. Any OTHER failure would + // leave stale rows that the re-extract then DUPLICATES (CodeRelation has no + // PK), so it must ABORT the writeback: re-throw so the caller's crash- + // recovery dirty flag forces a clean full rebuild on the next run. + const msg = err instanceof Error ? err.message : String(err); + if (/no table|not exist|not found|does not exist|Table .* does not exist/i.test(msg)) { + if (countResult) await closeQueryResults(countResult); + return { edgesDeleted }; + } + if (countResult) await closeQueryResults(countResult); + throw new Error( + `[call-summary] failed to clear existing CALL_SUMMARY edges before incremental ` + + `re-write (${msg}) — aborting to avoid duplicate summaries; ` + + `the next run will full-rebuild`, + ); + } + if (countResult) await closeQueryResults(countResult); + return { edgesDeleted }; +}; + // ============================================================================ // Full-Text Search (FTS) Functions // ============================================================================ diff --git a/gitnexus/src/core/lbug/schema.ts b/gitnexus/src/core/lbug/schema.ts index 4c8cc7cc5..7fd19f5bf 100644 --- a/gitnexus/src/core/lbug/schema.ts +++ b/gitnexus/src/core/lbug/schema.ts @@ -194,6 +194,7 @@ CREATE NODE TABLE Route ( responseKeys STRING[], errorKeys STRING[], middleware STRING[], + method STRING, PRIMARY KEY (id) )`; @@ -235,6 +236,8 @@ CREATE NODE TABLE BasicBlock ( startLine INT64, endLine INT64, text STRING, + callees STRING, + calleeIds STRING, PRIMARY KEY (id) )`; diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index ed92c9b9c..cf22cf09e 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -25,6 +25,7 @@ import { deleteNodesForFile, deleteAllCommunitiesAndProcesses, deleteAllInterprocTaintPaths, + deleteAllCallSummaries, queryImporters, loadFTSExtension, } from './lbug/lbug-adapter.js'; @@ -437,6 +438,13 @@ export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] => // writeback that recomputes the fuller coverage (no `--force` needed). // Bump this tag on any future change to which facts the solver emits. reachingDefSolver: 'ssa-sparse-v1', + // PDG FU-C: this run records CALL_SUMMARY return-value-ascent edges. + // Absent on any pre-FU-C (v3) stamp → the key-union pdgModeMismatch trips + // the first FU-C-aware run over an existing `--pdg` index and forces the + // full writeback that materialises CALL_SUMMARY edges without `--force`; + // and `impact`'s PDG mode reads its absence to note "no return-value + // ascent (re-index for CALL_SUMMARY)" on a v3 index (intra slice intact). + hasCallSummary: true, } : undefined; @@ -505,6 +513,14 @@ export const pdgModeMismatch = (recorded: RepoMeta['pdg'], options: PdgOptions): // a full writeback that populates REACHING_DEF rows without `--force`. const reqRecord = requested as Record; const recRecord = recorded as Record; + // INVARIANT: every value stamped by resolvePdgConfig MUST be a SCALAR (string / + // number / boolean). This comparison is a shallow `!==`, so an OBJECT or ARRAY + // value would compare by REFERENCE — two structurally-equal values from + // different runs would always be `!==`, tripping pdgModeMismatch on every + // re-analyze and forcing a needless full writeback. e.g. do NOT change + // `hasCallSummary: true` to a per-language object like `{ ts: true, ... }`; keep + // the diagnostic per-language refinement in the impact CONSUMER (see + // pdg-impact.ts assemblePdgImpactResult), not in this version discriminator. for (const key of new Set([...Object.keys(reqRecord), ...Object.keys(recRecord)])) { if (reqRecord[key] !== recRecord[key]) return true; } @@ -1116,6 +1132,12 @@ export async function runFullAnalysis( // graph (isGraphWideRelType), mirroring Community/Process. if (options.pdg === true) { await deleteAllInterprocTaintPaths(); + // 2c. Drop CALL_SUMMARY edges (PDG FU-C) on an incremental `--pdg` + // writeback. They are re-included from the FULL fresh graph + // (isGraphWideRelType) and the callSummaries phase recomputes every + // summary each run, so delete-all-then-rebuild keeps an unchanged + // function's summary from being lost — same contract as TAINT_PATH. + await deleteAllCallSummaries(); } // 3. Extract the changed subgraph from the FULL ctx.graph and write diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index c20433670..42edd119e 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -64,7 +64,29 @@ import { } from '../tools.js'; import { findImportCycles } from '../../core/graph/import-cycles.js'; import { decodeTaintPath } from '../../core/ingestion/taint/path-codec.js'; +import { decodeReachingDefReason } from '../../core/ingestion/cfg/reaching-def-reason-codec.js'; import { EXTENSIONS } from '../../core/ingestion/import-resolvers/utils.js'; +import { + fnLineOf, + isPdgDegradedLayerStatus, + makePdgImpactErrorResult, + makePdgLayerDegradedResult, + pdgLayerStatus, + pdgStampForMode, + runImpactPDG, + validateImpactMode, + pdgBridgeEvidenceForImpact, + betterBridgeEvidence, + composeUnifiedPdgImpactResult, + splitCalleeIds, + type ImpactMode, + type PdgImpactResult, + type PdgImpactErrorResult, + type PdgImpactTarget, + type PdgBridgeOptions, + type PdgBridgeEvidenceInfo, + type PdgLayerStatus, +} from './pdg-impact.js'; /** Real source-file extensions (`.ts`, `.py`, …) from the resolver's list, * excluding the empty entry and the `/index.*` forms — used to decide whether @@ -96,6 +118,7 @@ function resolveAliasString(canonical: unknown, legacy: unknown): string | undef } return undefined; } + // AI context generation is CLI-only (gitnexus analyze) // import { generateAIContextFiles } from '../../cli/ai-context.js'; @@ -425,7 +448,24 @@ interface ImpactParams { file_path?: string; kind?: string; direction: 'upstream' | 'downstream'; + /** + * Blast-radius engine (KTD1/KTD5). Absent / `undefined` / `'callgraph'` → + * the unchanged inter-procedural symbol→symbol BFS. `'pdg'` → the opt-in, + * intra-procedural Program Dependence Graph traversal (`_runImpactPDG`). + * Validated in `_impactImpl`; any other value is a hard `{ error }`. + */ + mode?: ImpactMode; + /** + * Statement anchor for `mode:'pdg'` (1-based source line). When provided, the + * PDG traversal seeds the dependence slice on the BasicBlock(s) at THIS line + * within the target symbol — answering "what statements depend on the code at + * line N?" — instead of the whole-symbol seed (which is empty for a function, + * since its intra-procedural reach stays inside its own blocks). Only + * meaningful with `mode:'pdg'`; rejected for `mode:'callgraph'`. + */ + line?: number; maxDepth?: number; + crossDepth?: number; relationTypes?: string[]; includeTests?: boolean; minConfidence?: number; @@ -2830,7 +2870,7 @@ export class LocalBackend { private async resolveBlockAnchor( repo: RepoHandle, target: string, - toolName: 'explain' | 'pdg_query', + toolName: 'explain' | 'pdg_query' | 'impact', ): Promise<{ anchorClause: string; queryParams: Record; @@ -3288,18 +3328,13 @@ export class LocalBackend { // Cheap meta probe: the layer exists iff the pdg stamp carries the // mode-relevant cap (maxCdgEdgesPerFunction for CDG, maxReachingDef… // for REACHING_DEF). Absent ⇒ the no-layer hint without a DB scan. - let pdgStamped: boolean | undefined; - try { - const meta = await loadMeta(path.dirname(repo.lbugPath)); - if (meta) { - pdgStamped = - mode === 'controls' - ? meta.pdg?.maxCdgEdgesPerFunction !== undefined - : meta.pdg?.maxReachingDefEdgesPerFunction !== undefined; - } - } catch { - /* meta unreadable — decide from the DB below */ - } + // `pdgStampForMode` is the shared meta read (the both-caps `pdgLayerStatus` + // helper consumes the same underlying read for impact); here we project it + // down to this one mode's cap, preserving the tri-state `boolean | undefined` + // contract byte-for-byte: `false` ⇒ definitive no-layer (short-circuit + // below), `true` ⇒ proceed, `undefined` ⇒ meta unreadable, defer to the + // post-anchored-query probe (Feasibility Issue 4). + const pdgStamped = await pdgStampForMode(repo.lbugPath, mode); if (pdgStamped === false) { return { mode, results: [], total: 0, note: NO_PDG_NOTE }; } @@ -3315,11 +3350,17 @@ export class LocalBackend { const queryParams = resolved.queryParams; // Optional variable filter (flows mode) — REACHING_DEF stores the variable - // name in `reason`. + // name in `reason`. FU-B-2 prefixes the name with a `|1::` + // annotation (name FIRST), so match BOTH a legacy bare-name reason (`=`) AND + // an annotated one (`STARTS WITH |`). Source identifiers never contain + // `|`, so the `|` prefix is exact (it cannot collide with a longer name + // — `ab|…` is not a prefix of `abc|…`). let reasonClause = ''; if (mode === 'flows' && typeof params.variable === 'string' && params.variable.trim()) { - reasonClause = ' AND r.reason = $variable'; - queryParams.variable = params.variable.trim(); + reasonClause = ' AND (r.reason = $variable OR r.reason STARTS WITH $variablePrefix)'; + const variable = params.variable.trim(); + queryParams.variable = variable; + queryParams.variablePrefix = `${variable}|`; } // edgeType is a hardcoded per-mode literal (never user input); `target` / @@ -3359,12 +3400,7 @@ export class LocalBackend { } // basicBlockId = `BasicBlock::::` — split - // from the RIGHT (filePath may contain ':'). - const fnLineOf = (id: string): number => { - const parts = id.split(':'); - return Number(parts[parts.length - 3]); - }; - + // from the RIGHT (filePath may contain ':'). Shared module-scope `fnLineOf`. const results = mode === 'controls' ? rows.map((r: any) => { @@ -3383,9 +3419,13 @@ export class LocalBackend { }) : rows.map((r: any) => { const fnLine = fnLineOf(String(r.srcId ?? r[0] ?? '')); + // FU-B-2: REACHING_DEF `reason` is `` (legacy) or + // `|1::` — decode to surface the bare + // variable name, not the encoded annotation. + const variable = decodeReachingDefReason(r.reason ?? r[4] ?? '').name; return { ...(Number.isInteger(fnLine) ? { functionLine: fnLine } : {}), - variable: String(r.reason ?? r[4] ?? ''), + variable, def: { line: (r.srcLine ?? r[1]) as number | undefined }, use: { line: (r.dstLine ?? r[2]) as number | undefined, @@ -4222,14 +4262,34 @@ export class LocalBackend { return await this._impactImpl(repo, params); } catch (err: any) { // Return structured error instead of crashing (#321) + const message = + (err instanceof Error ? err.message : String(err)) || 'Impact analysis failed'; + const suggestion = 'The graph query failed — try gitnexus context as a fallback'; + const recoverySuggestion = isWalCorruptionError(err) ? WAL_RECOVERY_SUGGESTION : undefined; + if (params.mode === 'pdg') { + // Symbol resolution never reached the catch with a resolved symbol (the + // throw can originate before/within resolution), so the envelope carries + // the partial-but-typed target — typed as PdgImpactTarget so the partial + // is type-checked, not an inline literal in a Promise hole. + const target: PdgImpactTarget = { name: params.target }; + const pdgErr: PdgImpactErrorResult = makePdgImpactErrorResult({ + mode: 'pdg', + error: message, + target, + direction: params.direction, + suggestion, + recoverySuggestion, + }); + return pdgErr; + } return { - error: (err instanceof Error ? err.message : String(err)) || 'Impact analysis failed', + error: message, target: { name: params.target }, direction: params.direction, impactedCount: 0, risk: 'UNKNOWN', - suggestion: 'The graph query failed — try gitnexus context as a fallback', - ...(isWalCorruptionError(err) ? { recoverySuggestion: WAL_RECOVERY_SUGGESTION } : {}), + suggestion, + ...(recoverySuggestion ? { recoverySuggestion } : {}), }; } } @@ -4238,6 +4298,82 @@ export class LocalBackend { await this.ensureInitialized(repo); const { target, direction } = params; + + // ── Dispatch order (KTD5) ────────────────────────────────────────── + // (1) Validate `mode`. Absent/'callgraph' → unchanged path; 'pdg' → the + // intra-procedural PDG engine; anything else → hard error. + // This MUST come before resolveSymbolCandidates so the ambiguous branch can + // fork on the validated mode and never run the callgraph fan-out under pdg. + const modeResult = validateImpactMode(params.mode); + if ('error' in modeResult) { + return { + error: modeResult.error, + target: { name: target }, + direction, + impactedCount: 0, + risk: 'UNKNOWN', + }; + } + const mode = modeResult.mode; + + // `line` is a PDG-only statement anchor. Reject it on the callgraph path + // rather than silently ignore (the symbol→symbol BFS has no statement notion). + if (params.line !== undefined && mode !== 'pdg') { + return { + error: `Parameter 'line' is only supported with mode:'pdg' (it anchors the dependence slice on a statement). Remove it or set mode:'pdg'.`, + target: { name: params.target }, + direction: params.direction, + impactedCount: 0, + risk: 'UNKNOWN', + }; + } + // A provided `line` must be a positive integer. + if ( + params.line !== undefined && + (!Number.isInteger(params.line) || (params.line as number) < 1) + ) { + // Line param fails validation before target resolution → partial-but-typed + // target on the pdg path (typed PdgImpactTarget, not an inline literal). + const badLineTarget: PdgImpactTarget = { name: params.target }; + return mode === 'pdg' + ? makePdgImpactErrorResult({ + mode: 'pdg', + error: `Parameter 'line' must be a positive integer (1-based source line), got ${JSON.stringify(params.line)}.`, + target: badLineTarget, + direction: params.direction, + }) + : { + error: `Parameter 'line' must be a positive integer (1-based source line), got ${JSON.stringify(params.line)}.`, + target: { name: params.target }, + direction: params.direction, + impactedCount: 0, + risk: 'UNKNOWN', + }; + } + + if (mode === 'pdg') { + // PDG mode is now unified inside a single repo: it combines the local + // CDG/RD statement slice with the same inter-symbol reach used for the + // option-driven comparison path. Cross-repo fan-out remains a callgraph + // feature, so crossDepth is still a loud error rather than a silent ignore. + const incompatible: string[] = []; + if (params.crossDepth !== undefined) incompatible.push('crossDepth'); + if (incompatible.length > 0) { + // crossDepth is rejected before target resolution → partial-but-typed + // target (typed PdgImpactTarget). + const crossDepthTarget: PdgImpactTarget = { name: target }; + const pdgErr: PdgImpactErrorResult = makePdgImpactErrorResult({ + mode: 'pdg', + error: + `Parameter(s) ${incompatible.join(', ')} are not supported with mode:'pdg' ` + + `(single-repo PDG impact). Remove them or use mode:'callgraph' for cross-repo fan-out.`, + target: crossDepthTarget, + direction, + }); + return pdgErr; + } + } + const maxDepth = params.maxDepth || 3; // Map legacy relation type names before filtering (backward compat for OVERRIDES → METHOD_OVERRIDES) const mappedRelTypes = params.relationTypes?.flatMap((t: string) => @@ -4290,16 +4426,67 @@ export class LocalBackend { if (outcome.kind === 'not_found') { const missing = params.target_uid ?? target; - return { - error: `Target '${missing}' not found`, - target: { name: target }, - direction, - impactedCount: 0, - risk: 'UNKNOWN', - }; + // not_found = no resolved symbol, so the envelope keeps the partial-but- + // typed target (typed PdgImpactTarget — there is no id/type/filePath yet). + const notFoundTarget: PdgImpactTarget = { name: target }; + return mode === 'pdg' + ? makePdgImpactErrorResult({ + mode: 'pdg', + error: `Target '${missing}' not found`, + target: notFoundTarget, + direction, + }) + : { + error: `Target '${missing}' not found`, + target: { name: target }, + direction, + impactedCount: 0, + risk: 'UNKNOWN', + }; } if (outcome.kind === 'ambiguous') { + // Shared truncation cap for the ambiguous candidate list — both the pdg + // branch (shows candidates) and the callgraph branch (probes candidates) + // bound to this many. + const AMBIGUOUS_MAX_CANDIDATES = 6; + // KTD5 ambiguous trap — under mode:'pdg' we MUST NOT fall into the + // callgraph fan-out below: it runs `_runImpactBFS` per candidate, which + // would silently execute the call-graph engine under a `pdg` call (the + // exact silent fallback KTD5 forbids). For U1 the pdg ambiguous path + // returns the candidate list WITHOUT any callgraph probe; the full pdg + // ambiguous handling (per-candidate PDG summaries / ranking) lands in U4. + if (mode === 'pdg') { + const truncated = outcome.candidates.length > AMBIGUOUS_MAX_CANDIDATES; + const shown = outcome.candidates.slice(0, AMBIGUOUS_MAX_CANDIDATES); + return { + status: 'ambiguous', + mode, + message: + `Found ${outcome.candidates.length} symbols matching '${target}'` + + (truncated ? ` (showing ${shown.length} of ${outcome.candidates.length})` : '') + + `. Disambiguate with target_uid (or file_path/kind) for a single ` + + `authoritative PDG result.`, + target: { name: target }, + direction, + totalCandidates: outcome.candidates.length, + // No single resolved symbol → impactedCount stays 0 / risk UNKNOWN + // (UNKNOWN must never read as "safe to refactor"). No callgraph + // fan-out runs, so there is no per-candidate blast radius here yet. + impactedCount: 0, + risk: 'UNKNOWN', + ...(truncated && { candidatesTruncated: true }), + candidates: shown.map((c) => ({ + uid: c.id, + name: c.name, + kind: c.type, + filePath: c.filePath, + line: c.startLine, + score: Number(c.score.toFixed(2)), + })), + }; + } + // #2129 — a bare name that collides with several symbols must NOT report a // bare `impactedCount: 0`. The real blast radius lives under whichever // candidate the caller meant; a flat zero here is precisely the silent @@ -4309,7 +4496,6 @@ export class LocalBackend { // summary-only BFS per candidate so each one's true count + risk is // visible, and surface the maximum at the top level so the headline can // never read as "safe to refactor". Candidates arrive sorted by score. - const AMBIGUOUS_MAX_CANDIDATES = 6; const probed = outcome.candidates.slice(0, AMBIGUOUS_MAX_CANDIDATES); // `partialProbe` is intentionally a SECOND incompleteness flag, distinct // from the traversal-interrupted `partial` flag used elsewhere: it means @@ -4422,9 +4608,49 @@ export class LocalBackend { id: outcome.symbol.id, name: outcome.symbol.name, filePath: outcome.symbol.filePath, + // Carry the resolved span so the PDG seed anchors on THIS symbol directly, + // without re-resolving its (possibly ambiguous) name (FIX 1). + startLine: outcome.symbol.startLine, + endLine: outcome.symbol.endLine, }; const symType = outcome.resolvedLabel || outcome.symbol.type || ''; + // (2) PDG-layer presence probe (U2, KTD7) — the four-state degradation + // contract after target resolution, before traversal. A repo never analyzed + // with `--pdg` (no-layer), one with only a partial layer (sub-layer-missing + // — impact needs BOTH CDG and REACHING_DEF), or one whose meta is unreadable + // (unknown) each returns a distinct guidance note here rather than a + // confusing empty blast radius. Resolve first so target-known degraded + // responses keep the same id/type/filePath envelope as successful PDG + // responses; only `ready` falls through to traversal. + // Hoisted so the (4) traversal branch can read `layer.hasCallSummary` (FU-C): + // the same single meta-stamp probe serves both the degradation gate and the + // ascent-availability note — no second probe. + let layer: PdgLayerStatus | undefined; + if (mode === 'pdg') { + layer = await pdgLayerStatus({ + lbugPath: repo.lbugPath, + executeParameterized, + }); + if (isPdgDegradedLayerStatus(layer)) { + // Degradation occurs AFTER target resolution → thread the FULL typed + // envelope ({ id, name, type, filePath }) so degraded responses keep the + // same target shape as a successful PDG result (typed PdgImpactTarget). + const degradedTarget: PdgImpactTarget = { + id: sym.id, + name: sym.name, + type: symType || 'Function', + filePath: sym.filePath, + }; + return makePdgLayerDegradedResult({ + mode, + layer, + target: degradedTarget, + direction, + }); + } + } + const effectiveRelationTypes = (symType === 'Class' || symType === 'Interface') && !hasExplicitRelationTypes && @@ -4432,6 +4658,86 @@ export class LocalBackend { ? [...relationTypes, 'ACCESSES'] : relationTypes; + // (4) single → route the resolved symbol to the engine selected by `mode`. + if (mode === 'pdg') { + const pdgResult = await this._runImpactPDG({ + repo, + sym, + symType, + direction, + maxDepth, + line: params.line, + limit: Number.isFinite(params.limit) ? params.limit : 100, + // KTD2 extraction-seam discipline: hand the engine its DB dependency + // explicitly rather than `this.`-binding it. LocalBackend owns repo + // lifecycle; `pdg-impact.ts` owns traversal/projection. + executeParameterized, + // FU-C: thread the CALL_SUMMARY layer presence read above (meta stamp) so + // the engine notes "no return-value ascent (re-index)" on a pre-FU-C (v3) + // index. `layer.hasCallSummary` is set on every meta-readable state. + callSummaryAvailable: layer?.hasCallSummary === true, + }); + + // Statement-precise inter-procedural reach: a first-hop callee is "proven" + // iff it is invoked in a block of the criterion's dependence slice. The + // slice = the seed block(s) (the changed line itself) UNION the dependent + // reachable blocks — both carry the leaf callee names they call + // (`BasicBlock.callees`). The seed block is included because a callee + // invoked directly on the changed line is the most-directly-impacted one, + // yet `reachableBlocks` excludes the seed by the seed-minus-reachable + // convention. Upstream seeds carry no discriminating slice, so the bridge + // falls back to preserving callgraph reach. + // `_runImpactPDG` returns the PdgImpactResult union; only the success/empty + // slice results carry intraReachableBlocks/seedBlocks (degraded and error + // results do not). Narrow via the same discriminant the composer uses, then + // read the typed string[] slices — no `as any`. + const sliceResult = 'error' in pdgResult || 'pdgLayer' in pdgResult ? null : pdgResult; + // FIX 6: key the "first-hop proven" set on the INTRA-procedural slice only + // (seed ∪ intra-reachable), NOT `reachableBlocks` — which the U1 descent now + // EXPANDS with inter-procedurally-reached callee blocks. Using the expanded + // superset would mark transitively-reached (2+ hop) callgraph targets as + // first-hop "proven", silently shifting the established statementPrecision + // semantics. The interproc-reached blocks are routed into the statement + // slice / block→symbol projection inside `_runImpactPDG` only. + const intraReachableBlocks: string[] = sliceResult?.intraReachableBlocks ?? []; + const seedBlocks: string[] = sliceResult?.seedBlocks ?? []; + const sliceBlocks = [...seedBlocks, ...intraReachableBlocks]; + const sliceCalleeNames = + direction === 'downstream' && sliceBlocks.length > 0 + ? await this.calleesOfBlocks(repo, sliceBlocks) + : new Set(); + // Resolved-id slice set (sound primary key, KTD3): unioned from the SAME + // seed ∪ reachable block set as the names — so a callee invoked only on the + // seeded line is still provable by id. Absent on a pre-v3 index (no + // `calleeIds` column) → empty → the bridge falls back to the leaf-name match. + const sliceCalleeIds = + direction === 'downstream' && sliceBlocks.length > 0 + ? await this.calleeIdsOfBlocks(repo, sliceBlocks) + : new Set(); + // Build the bridge when EITHER the name fallback or the id key has signal — + // an id-only index (names empty but ids present) must still seed the bridge. + const pdgBridge: PdgBridgeOptions | undefined = + sliceCalleeNames.size > 0 || sliceCalleeIds.size > 0 + ? { sliceCalleeNames, sliceCalleeIds } + : undefined; + + try { + const interproceduralResult = await this._runImpactBFS(repo, sym, symType, direction, { + maxDepth, + relationTypes: effectiveRelationTypes, + includeTests, + minConfidence, + limit: Number.isFinite(params.limit) ? params.limit : 100, + offset: Number.isFinite(params.offset) ? params.offset : 0, + pdgBridge, + }); + return composeUnifiedPdgImpactResult(pdgResult, interproceduralResult); + } catch (e) { + logQueryError('impact:pdg-interprocedural-reach', e); + return composeUnifiedPdgImpactResult(pdgResult, null, e); + } + } + return this._runImpactBFS(repo, sym, symType, direction, { maxDepth, relationTypes: effectiveRelationTypes, @@ -4443,6 +4749,93 @@ export class LocalBackend { }); } + /** + * Union of the leaf callee names invoked across a set of dependence-slice + * blocks (`BasicBlock.callees`, space-joined at emit). Drives statement-precise + * inter-procedural evidence: a first-hop callee reached from the criterion is + * "proven" (callgraph-bridge) iff its name is in this set, else unproven-bridge. + * Empty when the slice blocks call nothing or carry no harvested callees + * (non-TS/JS or synthetic ENTRY/EXIT blocks) — the bridge then preserves + * callgraph reach. A query failure is logged and degrades to empty (no proof), + * never throws (the inter-procedural reach is still returned). + */ + private async calleesOfBlocks(repo: RepoHandle, blockIds: string[]): Promise> { + const names = new Set(); + if (blockIds.length === 0) return names; + try { + const rows = await executeParameterized( + repo.lbugPath, + `MATCH (b:BasicBlock) WHERE b.id IN $ids RETURN b.callees AS callees`, + { ids: blockIds }, + ); + for (const r of rows as any[]) { + const raw = String(r.callees ?? r[0] ?? ''); + for (const n of raw.split(' ')) if (n) names.add(n); + } + } catch (e) { + logQueryError('impact:pdg-slice-callees', e); + } + return names; + } + + /** + * Union of the RESOLVED callee symbol ids invoked across a set of + * dependence-slice blocks (`BasicBlock.calleeIds`, space-joined at emit — + * sibling of `callees`). This is the SOUND key the bridge prefers: a first-hop + * callee is proven statement-precise iff its resolved id is in this set, which + * eliminates the same-leaf-name collision (false-positive) and import-alias + * (false-negative) the name set cannot distinguish. Empty when the slice blocks + * carry no captured ids (pre-v3 index without the `calleeIds` column, or + * non-overloading/synthetic blocks) — the bridge then falls back to the + * leaf-name match per U5. A query failure is logged and degrades to empty (no + * proof), never throws (the inter-procedural reach is still returned). Mirrors + * `calleesOfBlocks` exactly — same shape, same swallow-on-error contract. + */ + private async calleeIdsOfBlocks(repo: RepoHandle, blockIds: string[]): Promise> { + const ids = new Set(); + if (blockIds.length === 0) return ids; + try { + const rows = await executeParameterized( + repo.lbugPath, + `MATCH (b:BasicBlock) WHERE b.id IN $ids RETURN b.calleeIds AS calleeIds`, + { ids: blockIds }, + ); + for (const r of rows) { + // Shared split-and-drop-sentinel logic (`splitCalleeIds`) so this bridge + // key and the inter-procedural descent cannot diverge. The sentinel marks + // a capped block (handled by the names-sentinel check in the bridge) and + // is not a resolved symbol id, so it never enters the `has(realId)` set. + for (const id of splitCalleeIds(r.calleeIds ?? r[0])) ids.add(id); + } + } catch (e) { + logQueryError('impact:pdg-slice-callee-ids', e); + } + return ids; + } + + /** + * Delegates the PDG impact engine to `pdg-impact.ts`. + * + * The private method remains as the LocalBackend dispatch seam so existing + * tests can keep asserting that `mode:'pdg'` routes through the PDG + * statement engine before LocalBackend attaches interprocedural symbol reach. + * The traversal/projection/result assembly lives in the extracted helper + * module. + */ + private async _runImpactPDG(deps: { + repo: RepoHandle; + sym: { id: string; name: string; filePath: string; startLine?: number; endLine?: number }; + symType: string; + direction: 'upstream' | 'downstream'; + maxDepth: number; + limit: number; + line?: number; + executeParameterized: typeof executeParameterized; + callSummaryAvailable?: boolean; + }): Promise { + return runImpactPDG(deps); + } + /** * #1858 — epistemic lower-bound detection. * @@ -4592,6 +4985,7 @@ export class LocalBackend { skipPerSymbolEnrichment?: boolean; skipEpistemic?: boolean; skipEnrichment?: boolean; + pdgBridge?: PdgBridgeOptions; }, ): Promise { const { maxDepth, relationTypes, includeTests, minConfidence } = opts; @@ -4634,6 +5028,7 @@ export class LocalBackend { const impacted: any[] = []; const visited = new Set([symId]); + const pdgBridgeEvidenceById = new Map(); let frontier = [symId]; let traversalComplete = true; @@ -4731,11 +5126,36 @@ export class LocalBackend { }); for (const rel of related) { + const sourceId = String(rel.sourceId ?? rel[0] ?? ''); const relId = rel.id || rel[1]; const filePath = rel.filePath || rel[4] || ''; if (!includeTests && isTestFilePath(filePath)) continue; + // Bridge evidence is computed for EVERY edge (not just the first to + // reach a node) and the strongest verdict across all parents is kept + // (`callgraph-bridge` wins). This makes a diamond-reachable node's + // proven/unproven label order-independent of DB row iteration; the + // final label is stamped onto the impacted items after the depth loop. + if (opts.pdgBridge) { + const ev = pdgBridgeEvidenceForImpact({ + bridge: opts.pdgBridge, + depth, + calleeName: rel.name || rel[2], + // Sound primary key (KTD3): the reached callee's RESOLVED id — the + // same `relId` (`rel.id`) the BFS keys its visited/frontier sets on, + // which equals the CALLS targetId captured into `BasicBlock.calleeIds`. + // The bridge proves by id ∈ `sliceCalleeIds` first, falling back to + // `calleeName` only when ids are absent or the block is capped. + calleeId: relId, + inherited: pdgBridgeEvidenceById.get(sourceId), + }); + pdgBridgeEvidenceById.set( + String(relId), + betterBridgeEvidence(pdgBridgeEvidenceById.get(String(relId)), ev), + ); + } + if (!visited.has(relId)) { visited.add(relId); nextFrontier.push(relId); @@ -4747,6 +5167,8 @@ export class LocalBackend { typeof storedConfidence === 'number' && storedConfidence > 0 ? storedConfidence : confidenceForRelType(relationType); + // pdgEvidence is stamped after the depth loop from the finalized, + // order-independent pdgBridgeEvidenceById map. impacted.push({ depth, id: relId, @@ -4769,6 +5191,19 @@ export class LocalBackend { frontier = nextFrontier; } + // Stamp the finalized, order-independent bridge evidence (strongest across + // all parents) onto each impacted item. Deferred from the BFS loop so a + // diamond-reachable node reflects a proven parent regardless of visit order. + if (opts.pdgBridge) { + for (const item of impacted as Array>) { + const ev = pdgBridgeEvidenceById.get(String(item.id)); + if (ev) { + item.pdgEvidence = ev.evidence; + item.pdgBridgeBasis = ev.basis; + } + } + } + const grouped: Record = {}; for (const item of impacted) { if (!grouped[item.depth]) grouped[item.depth] = []; @@ -5348,6 +5783,32 @@ export class LocalBackend { const svc = this.getGroupService(); if (method === 'impact') { + // KTD5/KTD12 — validate `mode` at the group-forward boundary too (the + // JSON-schema enum is advisory). An invalid mode errors; `mode:'pdg'` is + // rejected for @group targets because PDG impact is single-repo and + // intra-procedural — there is no cross-repo dependence graph to walk. + // Rejecting here (before groupImpact) is the KTD12 @group hard error. + const groupModeResult = validateImpactMode(params.mode); + if ('error' in groupModeResult) return { error: groupModeResult.error }; + if (groupModeResult.mode === 'pdg') { + // @group reject: no single-repo symbol is ever resolved on the group path, + // so the envelope carries the partial-but-typed target (PdgImpactTarget). + // Routed through the typed builder so this exit is a PdgImpactResult union + // member, never a bare { error } object. + const groupRejectTarget: PdgImpactTarget = { name: String(params.target ?? '') }; + const pdgErr: PdgImpactErrorResult = makePdgImpactErrorResult({ + mode: 'pdg', + error: + "mode:'pdg' is not supported for @group targets — PDG impact is " + + 'single-repo and intra-procedural. Run pdg impact against an ' + + 'individual indexed repository instead.', + target: groupRejectTarget, + direction: (params.direction === 'downstream' ? 'downstream' : 'upstream') as + | 'upstream' + | 'downstream', + }); + return pdgErr; + } const impactArgs: Record = { name: groupName, repo: resolved.repoPath, diff --git a/gitnexus/src/mcp/local/pdg-impact.ts b/gitnexus/src/mcp/local/pdg-impact.ts new file mode 100644 index 000000000..434d03a5a --- /dev/null +++ b/gitnexus/src/mcp/local/pdg-impact.ts @@ -0,0 +1,2616 @@ +/** + * PDG-backed impact helpers. + * + * Extracted from `local-backend.ts` so LocalBackend owns dispatch/repo lifecycle + * while this module owns the PDG layer probe, statement traversal, block + * projection, and result assembly contract. + */ + +import path from 'path'; +import type { executeParameterized } from '../../core/lbug/pool-adapter.js'; +import { loadMeta } from '../../storage/repo-manager.js'; +import { IMPACT_MAX_DEPTH, PDG_QUERY_DEFAULT_LIMIT, PDG_QUERY_MAX_LIMIT } from '../tools.js'; +import { CALLEES_TRUNCATED_SENTINEL, CALLEE_ID_SEP } from '../../core/ingestion/cfg/emit.js'; +import { decodeCallSummary } from '../../core/ingestion/taint/call-summary-codec.js'; +import { decodeReachingDefReason } from '../../core/ingestion/cfg/reaching-def-reason-codec.js'; +import { getProviderForFile } from '../../core/ingestion/languages/index.js'; +import { SupportedLanguages } from 'gitnexus-shared'; + +/** + * Parse the `` segment out of a `BasicBlock` id (1-based function start + * line). The id template is + * `BasicBlock::::` + * and `` may itself contain `':'` (a Windows drive letter), so the + * segments are taken from the RIGHT: `` is last, `` second-last, + * `` third-last. Extracted from the `_pdgQueryImpl` closure (#2086) into + * a shared module-scope helper so the PDG impact traversal (U3/U4) reuses the + * exact same parse — the `pdg_query` read path is byte-identical to before. + */ +export function fnLineOf(id: string): number { + const parts = id.split(':'); + return Number(parts[parts.length - 3]); +} + +/** + * Parse the `` segment out of a `BasicBlock` id, the COUNTERPART to + * `fnLineOf`. The id template is `BasicBlock::::`, + * so the file path is everything BETWEEN the `BasicBlock:` prefix and the last + * THREE colon-segments (`::`). `` may itself + * contain `':'` (a Windows drive letter), so we strip from both ends rather than + * split-and-pick. Returns `''` for an unparseable id (treated as unresolved). + */ +function fnFileOf(id: string): string { + const parts = id.split(':'); + // Need: `BasicBlock` + filePath(≥1) + fnLine + fnCol + blockIdx ⇒ ≥5 segments. + if (parts.length < 5 || parts[0] !== 'BasicBlock') return ''; + // Drop the leading `BasicBlock` token and the trailing fnLine/fnCol/blockIdx, + // rejoin the middle on ':' to restore a path that itself contained colons. + return parts.slice(1, parts.length - 3).join(':'); +} + +/** + * Default inter-procedural FUNCTION-hop budget for the U1 forward closure (the + * number of call boundaries the slice descends through). Distinct from the per- + * hop intra-callee step/edge budget (which reuses `stepLimit`). Kept small (3) + * because context-insensitive descent over-includes with depth. + */ +const INTERPROC_DEPTH_BUDGET = 3; + +/** + * Total newly-reached-block cap across all inter-procedural hops — the secondary + * termination guard against a pathological fan-out inside the depth budget. + * Stamps `truncated` with the `'limit'` reason when it fires (it is a SIZE/budget + * cap, not dependence-level depth exhaustion — FIX 7). + */ +const INTERPROC_NODE_BUDGET = 5000; + +/** + * Split a tab-joined ({@link CALLEE_ID_SEP}) `BasicBlock.calleeIds` cell into its resolved callee + * symbol ids, dropping the truncation sentinel (a capped block carries the + * sentinel to mark an incomplete call-site list; it is NOT a resolved symbol id + * and must never enter a `has(realId)` set). Empty/whitespace cells yield no ids. + * + * Extracted here (U1) so the two callers — `LocalBackend.calleeIdsOfBlocks` (the + * statement-precise bridge key) and the inter-procedural descent's + * `calleeIdsFromCalleeRows` — cannot diverge on the split-and-drop-sentinel + * logic. Both consume rows of `BasicBlock.calleeIds`; this is the single source. + */ +export function splitCalleeIds(raw: unknown): string[] { + const out: string[] = []; + // Split on the SHARED CALLEE_ID_SEP (tab) — ids embed file paths / multi-word + // C++ type tokens that can contain a space, so a space split would fragment + // them. Producer (calleeIdsOfBlock) joins with the same constant. + for (const id of String(raw ?? '').split(CALLEE_ID_SEP)) { + if (id && id !== CALLEES_TRUNCATED_SENTINEL) out.push(id); + } + return out; +} + +/** + * Contract version of the mode:'pdg' impact result shape. A stable discriminator + * for external MCP/agent consumers — distinct from the DB INCREMENTAL_SCHEMA_VERSION. + * Bump on any breaking change to the PDG result fields. + */ +export const PDG_RESULT_VERSION = 1 as const; + +/** A reachable dependence block resolved to its source statement. */ +export interface PdgStatement { + /** 1-based source line where the statement's block starts. */ + line: number; + /** Repo-relative file path (parsed from the block id). */ + filePath: string; + /** The statement's source text (BasicBlock.text), trimmed. */ + text: string; + /** + * Whether the statement belongs to the criterion's OWN function (`'intra'`) + * or was reached across a call boundary (`'inter'`). A block is `'intra'` + * iff its owning-fn file AND 1-based owning-fn start line both match the + * criterion's — `fnFileOf(id) === criterionFile && fnLineOf(id) === ownerFnLine`. + * Projection-only tag (FU-A): NOT persisted, NOT a schema field. The bench + * intra axis scopes to `'intra'` statements so U1's cross-function reach + * stops being counted as intra-axis false positives; existing consumers + * ignore it. + */ + scope: 'intra' | 'inter'; +} + +/** + * FU-B-2 intra-block def→use line walk. Given a block's self REACHING_DEF + * def→use line PAIRS (every `defLine → useLine` step decoded from the edge + * `reason`'s pair list), walk FORWARD from a set of entry lines and return every + * interior USE line transitively reached. This is the principled statement- + * granular recovery for a coalesced straight-line BasicBlock, for BOTH chain + * shapes: + * - DISTINCT bindings: `chainCompute` coalesces `a@7; b@8; c@9` into one block; + * each binding is its own (block-pair, binding) group, so `a@7→b@8` and + * `b@8→c@9` arrive as separate self-edges, and walking from line 7 recovers + * {8, 9}. + * - SAME binding (reassignment): `acc = f(acc); acc = g(acc); acc = h(acc)` + * coalesces into one self-block whose `acc@24→acc@25`, `acc@25→acc@26`, + * `acc@26→acc@27` steps ALL share the one (self-block, accIdx) group — the + * dedup collapses them onto one edge, but the FU-B-2 pair LIST carries all + * three, so walking from line 24 chains 24→25→26→27 to fixpoint. (A + * first-pair-only annotation would stop at 25.) + * + * Pure: forward adjacency `defLine → Set`, BFS from `entryLines`, + * bounded by the (finite) pair set + a `reached` set so it always terminates, + * even on a cycle — a self-referential line (`x = x + 1`, def@L→use@L) re-adds + * nothing new. The entry lines themselves are NOT emitted (they are the seed / + * the block's representative line already surfaced elsewhere); only newly-reached + * interior use lines are. + */ +function walkIntraBlockChain( + selfEdges: ReadonlyArray<{ defLine: number; useLine: number }>, + entryLines: Iterable, +): Set { + const succ = new Map>(); + for (const { defLine, useLine } of selfEdges) { + if (!Number.isInteger(defLine) || !Number.isInteger(useLine)) continue; + const set = succ.get(defLine) ?? new Set(); + set.add(useLine); + succ.set(defLine, set); + } + const reached = new Set(); + const frontier: number[] = []; + for (const e of entryLines) frontier.push(e); + const seenSeed = new Set(frontier); + while (frontier.length > 0) { + const cur = frontier.pop() as number; + for (const next of succ.get(cur) ?? []) { + if (reached.has(next) || seenSeed.has(next)) continue; + reached.add(next); + frontier.push(next); + } + } + return reached; +} + +/** + * Fetch the self REACHING_DEF edges (`(a)-[REACHING_DEF]->(a)`) for a set of + * blocks, grouped by block id, with each edge's def/use source LINES decoded + * from the FU-B-2 `reason` annotation (`|1::`). A + * pre-FU-B-2 (un-annotated) `reason` decodes to no line info → the edge is + * dropped (the block then projects at block-start granularity exactly as before, + * the documented graceful degrade for an older index). A query error propagates. + */ +async function selfReachingDefEdgesByBlock( + lbugPath: string, + blockIds: string[], + exec: typeof executeParameterized, +): Promise>> { + const out = new Map>(); + if (blockIds.length === 0) return out; + const rows = await exec( + lbugPath, + `MATCH (a:BasicBlock)-[r:CodeRelation]->(a) + WHERE r.type = 'REACHING_DEF' AND a.id IN $ids + RETURN a.id AS id, r.reason AS reason`, + { ids: blockIds }, + ); + for (const r of rows as Array>) { + const id = String(r['id'] ?? ''); + if (!id) continue; + const decoded = decodeReachingDefReason(r['reason']); + // FU-B-2: the annotation carries the FULL ordered (defLine, useLine) pair + // list for this (block-pair, binding) group — push EVERY pair so the walk can + // chain a same-binding reassignment (`acc@24->acc@25->acc@26`), which the + // dedup coalesces onto this one edge. A pre-FU-B-2 (un-annotated) reason + // decodes to no pairs → contributes nothing (block-start granularity). + if (decoded.pairs.length === 0) continue; + const list = out.get(id) ?? []; + for (const p of decoded.pairs) list.push({ defLine: p.defLine, useLine: p.useLine }); + out.set(id, list); + } + return out; +} + +/** + * Resolve a set of BasicBlock ids to their source statements (line + text), + * deduped by `(filePath, line, block id)` and sorted by line. This is the useful + * output of a statement-anchored PDG slice — the dependent statements the change + * reaches. A query error propagates (no `.catch` swallow) so a DB failure is + * never silently reported as "no affected statements". + * + * FU-B-2 statement granularity: a coalesced straight-line BasicBlock is projected + * to its single `startLine` by default, which UNDER-reports the interior + * statements that genuinely depend on the criterion (lines 8/9 of a `7-9` block). + * For each block in `chainWalkBlocks` we walk its self REACHING_DEF def→use LINE + * chain (decoded from the FU-B-2 `reason` annotation) FORWARD from that block's + * entry line(s) and emit one statement per reached INTERIOR line (its own text + * from the block text). This is the SINGLE principled statement-granular + * mechanism — it replaces FU-C's blind all-interior-lines expansion of ascent + * blocks (now routed through the same walk, seeded at the block's start line) and + * also recovers the seed block's own interior dependents (the U2/intra-dataflow- + * chain recall gap). A pre-FU-B-2 index (no line annotation) yields no self-edge + * lines → the block degrades to its block-start projection (byte-identical to the + * old behavior). + */ +async function pdgStatementsForBlocks( + lbugPath: string, + blockIds: string[], + exec: typeof executeParameterized, + criterionFile: string, + ownerFnLine: number, + /** + * Blocks whose interior should be expanded to statement granularity via the + * self-edge def→use line walk, each mapped to the entry line(s) the walk + * starts from. Two contributors: + * - the SEED block, seeded at the criterion line — recovers the interior + * dependents of the changed statement inside its own coalesced block (the + * U2 intra-dataflow-chain gap: line 7's block 7-9 yields {8,9}); + * - each U-C4 ascent-confirmed CALL block, seeded at its start line — the + * statement-granular realisation of the return-value ascent (replacing the + * FU-C blind interior-line stop-gap with the principled walk). + * Empty/absent ⇒ every block projects at block-start granularity (byte- + * identical to the pre-FU-B-2 behavior). + */ + chainWalkBlocks: ReadonlyMap> = new Map(), +): Promise { + if (blockIds.length === 0 && chainWalkBlocks.size === 0) return []; + // The seed block(s) we walk are NOT in `blockIds` (seeds are excluded from the + // reachable slice), so the block-text fetch must cover BOTH the reachable + // blocks and the chain-walk blocks. Union, de-duplicated. + const fetchIds = [...new Set([...blockIds, ...chainWalkBlocks.keys()])]; + const [rows, selfEdgesByBlock] = await Promise.all([ + exec( + lbugPath, + `MATCH (b:BasicBlock) WHERE b.id IN $ids + RETURN b.id AS id, b.startLine AS line, b.endLine AS endLine, b.text AS text`, + { ids: fetchIds }, + ), + chainWalkBlocks.size > 0 + ? selfReachingDefEdgesByBlock(lbugPath, [...chainWalkBlocks.keys()], exec) + : Promise.resolve(new Map>()), + ]); + const byKey = new Map(); + const reachableIds = new Set(blockIds); + // Narrow the awaited rows ONCE at the boundary to a typed record shape; read + // the aliased cells via bracket access with String()/Number() coercion. + for (const r of rows as Array>) { + const id = String(r['id'] ?? ''); + const line = Number(r['line'] ?? 0); + if (!id || !Number.isFinite(line) || line <= 0) continue; + const filePath = fnFileOf(id); + const text = String(r['text'] ?? '').trim(); + // INTRA iff this block's owning function (file + 1-based start line) is the + // criterion's own function; otherwise it was reached across a call boundary + // (INTER). Pure key comparison parsed from the block id — no extra DB query. + const scope: 'intra' | 'inter' = + filePath === criterionFile && fnLineOf(id) === ownerFnLine ? 'intra' : 'inter'; + const textLines = String(r['text'] ?? '').split('\n'); + // ── FU-B-2 chain-walk block: expand to its interior dependent lines ─────── + // Walk the block's self REACHING_DEF def→use LINE chain forward from the + // entry line(s). Each reached interior line is a statement transitively + // data-dependent on the entry (the chain the coalesced block lost). Emit it + // with its own physical-line text (`BasicBlock.text` is `lines.join('\n')`, + // 1-based from `startLine`). This is the principled replacement for FU-C's + // blind all-interior expansion — only lines the def→use chain proves are + // surfaced. A block that is ALSO a reachable block keeps its block-start + // statement too (added below); the seed block (not reachable) contributes + // ONLY its walked interior lines. + const entryLines = chainWalkBlocks.get(id); + if (entryLines !== undefined) { + // An empty entry set means "seed from the block's own start line" — the + // FU-C ascent blocks, whose call statements chain forward from the block + // start (the caller does not know the start line at map-build time). A + // non-empty set (the seed block, seeded at the criterion line) is used + // verbatim. `line` is the resolved block start line for this row. + const seeds = entryLines.size > 0 ? entryLines : new Set([line]); + const reached = walkIntraBlockChain(selfEdgesByBlock.get(id) ?? [], seeds); + for (const ln of reached) { + const idx = ln - line; + if (idx < 0 || idx >= textLines.length) continue; // out of this block's text + const lineText = textLines[idx].trim(); + const key = `${filePath}:${ln}:${id}`; + if (!byKey.has(key)) byKey.set(key, { line: ln, filePath, text: lineText, scope }); + } + } + // A reachable block always surfaces its representative block-start statement. + // A pure chain-walk seed block (not reachable) does NOT (its start line is + // the criterion / the call block already surfaced elsewhere — only its + // walked-forward dependents are new). When the reachable block is ALSO a + // chain-walk block (a coalesced call block whose interior lines we expanded), + // its block-start statement is surfaced at the SAME statement granularity — + // the first physical line's own text, not the whole multi-statement block + // text — so all of its statements are consistently single-line. A reachable + // block that is NOT chain-walked keeps the full trimmed block text (byte- + // identical to the pre-FU-B-2 projection). + if (reachableIds.has(id)) { + const key = `${filePath}:${line}:${id}`; + const startText = + entryLines !== undefined && textLines.length > 0 ? textLines[0].trim() : text; + if (!byKey.has(key)) byKey.set(key, { line, filePath, text: startText, scope }); + } + } + return [...byKey.values()].sort((a, b) => { + if (a.filePath !== b.filePath) return a.filePath < b.filePath ? -1 : 1; + if (a.line !== b.line) return a.line - b.line; + return a.text < b.text ? -1 : a.text > b.text ? 1 : 0; + }); +} + +// ── Block → owning-symbol projection types (U4) ────────────────────────────── + +/** + * One owning-symbol candidate for a reachable BasicBlock, OR an explicit + * `unresolved` marker for a block that maps to no `Function`/`Method`/`Constructor` symbol + * (top-level/free-statement block, or a nested lambda whose start line ≠ any + * symbol `startLine`). A null `id` is the shadow-path marker — the block is + * surfaced under its file, never silently dropped (R9: a silent drop is a hidden + * recall loss). + */ +interface OwningSymbol { + /** Symbol UID, or `null` for the `unresolved` shadow-path entry. */ + id: string | null; + name: string; + /** `'Function' | 'Method' | …`, or `'unresolved'` for the shadow path. */ + type: string; + filePath: string; + /** Symbol `startLine` (0-based), present only for a resolved symbol. */ + startLine?: number; + /** + * True when this block's `(filePath, startLine)` query matched >1 symbol — + * same-line, different-name functions that the schema cannot disambiguate + * (no `startColumn` column; Feasibility Finding 1). ALL colliding symbols are + * reported (never a silent pick), each carrying this flag. + */ + ambiguous?: boolean; +} + +/** + * Net-new block → owning-symbol resolver (U4) — the REVERSE of + * `resolveBlockAnchor` (which goes symbol→blocks). No precedent exists: + * `_pdgQueryImpl` only ever extracts a raw `functionLine`, never an owning + * symbol. Lives in the extracted PDG impact engine and takes injected deps so + * LocalBackend keeps repo lifecycle/dispatch while this module owns projection. + * + * For each reachable block id `BasicBlock::::`: + * - `fnLineOf` → 1-based function start line; `fnFileOf` → file path. + * - Query `Function`/`Method`/`Constructor` `WHERE filePath = $f AND startLine = (fnLine-1)` + * — block `fnLine` is 1-based, symbol `startLine` is 0-based, so subtract one + * (the `[symStart+1]` convention from `resolveBlockAnchor`, applied in + * reverse; NOT re-derived). + * + * Two non-happy paths, BOTH surfaced (never silent): + * - **>1 match** (same-line different-name functions): `fnCol` rides the block + * id but the schema has NO `startColumn` column and the symbol id encodes only + * the name, so a `(filePath, startLine)` join cannot disambiguate. Report ALL + * colliding symbols, each `ambiguous: true` (R4 / Feasibility Finding 1). + * - **0 matches** (top-level/free-statement block, or a lambda whose start line + * ≠ a symbol `startLine`): one `unresolved` entry (`id: null`) under the + * block's file (R9 shadow path). + * + * Distinct `(filePath, fnLine)` pairs are queried once each (a block and its + * siblings in the same function share a pair), so the cost is O(distinct + * functions), not O(blocks). + */ +async function projectBlocksToSymbols(deps: { + lbugPath: string; + blockIds: string[]; + executeParameterized: typeof executeParameterized; +}): Promise<{ symbols: OwningSymbol[]; unresolvedCount: number; ambiguousCount: number }> { + const { lbugPath, blockIds, executeParameterized: exec } = deps; + + // Group blocks by their (filePath, fnLine) owning-function key so each owning + // function is resolved with a single query regardless of block count. + const byFnKey = new Map(); + for (const id of blockIds) { + const filePath = fnFileOf(id); + const fnLine = fnLineOf(id); // 1-based + if (!filePath || !Number.isFinite(fnLine)) { + // Unparseable block id — record an unresolved key so it is reported, never + // dropped. Use the raw id as the key so duplicates collapse. + byFnKey.set(`#bad#${id}`, { filePath: filePath || id, symStart: NaN }); + continue; + } + const symStart = fnLine - 1; // 0-based symbol startLine (reverse [symStart+1]) + byFnKey.set(`${filePath}#${symStart}`, { filePath, symStart }); + } + + const resolved: OwningSymbol[] = []; + let unresolvedCount = 0; + let ambiguousCount = 0; + + await Promise.all( + Array.from(byFnKey.values()).map(async ({ filePath, symStart }) => { + if (!Number.isFinite(symStart)) { + // Unparseable id — shadow-path unresolved under (best-effort) file. + resolved.push({ id: null, name: '(unresolved)', type: 'unresolved', filePath }); + unresolvedCount += 1; + return; + } + // `Function`/`Method`/`Constructor` carry name+filePath+startLine; the schema has NO + // `startColumn`, so the join is on (filePath, startLine) only. `filePath` + // and `symStart` are BOUND as params (KTD11 — never interpolated). A + // UNION ALL across explicit labels is used rather than a + // `(s:Function OR s:Method OR s:Constructor)` disjunction (unsupported in the LadybugDB + // Cypher subset — the established cross-label pattern, see + // `enrichCandidateLabels`). + // FIX 6: do NOT swallow a query failure as `[]`. A DB error (lock / + // corruption / missing path) must NOT masquerade as a genuine no-owning- + // symbol result — that would silently inflate `unresolvedCount` and hide + // the failure. Letting it reject propagates through `Promise.all` → + // `projectBlocksToSymbols` → `_runImpactPDG` → `_impactImpl` up to the + // `impact()` structured-error catch, where it surfaces as a real error + // with a recovery suggestion (rather than a clean-looking partial radius). + const rows = await exec( + lbugPath, + `MATCH (s:\`Function\`) + WHERE s.filePath = $filePath AND s.startLine = $symStart + RETURN s.id AS id, s.name AS name, 'Function' AS label, s.startLine AS startLine + UNION ALL + MATCH (s:\`Method\`) + WHERE s.filePath = $filePath AND s.startLine = $symStart + RETURN s.id AS id, s.name AS name, 'Method' AS label, s.startLine AS startLine + UNION ALL + MATCH (s:\`Constructor\`) + WHERE s.filePath = $filePath AND s.startLine = $symStart + RETURN s.id AS id, s.name AS name, 'Constructor' AS label, s.startLine AS startLine`, + { filePath, symStart }, + ); + + if (rows.length === 0) { + // No owning symbol — top-level/free-statement block or a lambda whose + // start line ≠ a symbol startLine. Shadow path: report under its file. + resolved.push({ + id: null, + name: '(unresolved)', + type: 'unresolved', + filePath, + startLine: symStart, + }); + unresolvedCount += 1; + return; + } + + // >1 ⇒ ambiguous-projection (same-line, different-name functions). Report + // ALL colliding symbols, NEVER silently pick one (R4 / Feasibility 1). + const isAmbiguous = rows.length > 1; + // Narrow the rows ONCE at the boundary to a typed record shape and read the + // aliased cells via bracket access (with positional `['0']`… fallback for a + // non-aliased row shape) — no per-field `as any`, matching the typed-row + // pattern used elsewhere in this file (e.g. lines ~264, ~1309, ~1386). + for (const r of rows as Array>) { + resolved.push({ + id: String(r['id'] ?? r['0'] ?? ''), + name: String(r['name'] ?? r['1'] ?? ''), + type: String(r['label'] ?? r['2'] ?? 'Function'), + filePath, + startLine: Number(r['startLine'] ?? r['3'] ?? symStart), + ...(isAmbiguous ? { ambiguous: true as const } : {}), + }); + } + if (isAmbiguous) ambiguousCount += 1; + }), + ); + + // Deterministic order: by filePath, then startLine, then id (unresolved last + // within a file). Order-independence matters for the parity/fingerprint + // contract (KTD8 standing interchangeability) and for stable consumer output. + resolved.sort((a, b) => { + if (a.filePath !== b.filePath) return a.filePath < b.filePath ? -1 : 1; + const al = a.startLine ?? Number.MAX_SAFE_INTEGER; + const bl = b.startLine ?? Number.MAX_SAFE_INTEGER; + if (al !== bl) return al - bl; + const ai = a.id ?? '￿'; + const bi = b.id ?? '￿'; + return ai < bi ? -1 : ai > bi ? 1 : 0; + }); + + return { symbols: resolved, unresolvedCount, ambiguousCount }; +} + +/** + * The KTD8 parity fields a PDG impact result carries even when it short-circuits + * to an empty radius (degraded layer / no PDG body / no dependence reachability). + * + * A programmatic consumer iterating `byDepth`, reading `byDepthCounts[1]`, or + * coalescing `affected_processes`/`affected_modules` must find a well-formed + * (empty) shape on EVERY early return, not `undefined` (which would render as + * "isolated"/"no data" instead of "inconclusive"). The CLI branches on + * `pdgLayer` first so it is safe regardless, but the JSON contract must be + * uniform across all three early returns — this single source guarantees that. + */ +function emptyPdgParityFields(): { + byDepth: Record; + byDepthCounts: Record; + summary: { direct: number; processes_affected: number; modules_affected: number }; + affected_processes: unknown[]; + affected_modules: unknown[]; +} { + return { + byDepth: {}, + byDepthCounts: { 1: 0 }, + summary: { direct: 0, processes_affected: 0, modules_affected: 0 }, + affected_processes: [], + affected_modules: [], + }; +} + +export interface PdgImpactTarget { + name: string; + id?: string; + type?: string; + filePath?: string; +} + +export interface PdgImpactParityFields { + byDepth: Record; + byDepthCounts: Record; + summary: { direct: number; processes_affected: number; modules_affected: number }; + affected_processes: unknown[]; + affected_modules: unknown[]; +} + +export type PdgImpactEvidence = + | 'local-dependence' + | 'owner-projection' + | 'callgraph-bridge' + | 'unproven-bridge' + | 'degraded'; + +export interface PdgImpactEvidenceSummary { + statements?: PdgImpactEvidence; + localSymbols?: PdgImpactEvidence; + interprocedural?: PdgImpactEvidence; + localSymbolCount?: number; + unresolvedBlockCount?: number; + ambiguousProjectionCount?: number; + interproceduralEvidenceCounts?: Partial>; +} + +export interface PdgInterproceduralImpact { + engine: 'symbol-graph'; + evidence: Extract; + impactedCount: number; + byDepthCounts: Record; + byDepth: Record; + evidenceCounts?: Partial>; + /** + * Statement-precise (proven) subset of `byDepth` — additive. Tighter than + * `byDepth` for a line-seeded downstream slice (drops `unproven-bridge` + * symbols not invoked from the dependence slice), equal to it otherwise. + * `statementPrecision` = |proven| / |reach| (null when there is no reach). + */ + statementPreciseByDepth?: Record; + statementPreciseByDepthCounts?: Record; + statementPreciseImpactedCount?: number; + statementPrecision?: number | null; + partial: boolean; +} + +export interface PdgImpactBaseResult extends PdgImpactParityFields { + mode: 'pdg'; + /** Contract version of the mode:'pdg' impact result shape; bump on any breaking change to the PDG result fields. */ + pdgResultVersion: 1; + target: PdgImpactTarget; + direction: 'upstream' | 'downstream'; + impactedCount: number; + risk: 'UNKNOWN'; + note?: string; + partial?: boolean; + interproceduralByDepth?: Record; + interproceduralByDepthCounts?: Record; + interproceduralEpistemic?: string; + interproceduralBoundaries?: unknown[]; + interproceduralError?: string; + // Statement-precise (proven) inter-procedural reach lives ONLY under + // `pdgInterprocedural` (the scoped namespace) — see PdgInterproceduralImpact. + pdgInterprocedural?: PdgInterproceduralImpact; + pdgEvidence?: PdgImpactEvidenceSummary; +} + +/** + * Slice-result fields shared verbatim by {@link PdgImpactSuccessResult} and + * {@link PdgImpactEmptyResult}. Those two differ ONLY in their `epistemic` + * discriminant (and the narrowed `target`, which must be re-declared on each to + * override `PdgImpactBaseResult.target`), so every other slice field lives here + * to keep the two in lockstep — a new slice field is added once, not twice. + */ +export interface PdgImpactSliceFields { + reachableBlocks: string[]; + /** + * INTRA-procedural reachable subset of `reachableBlocks` (the statement slice + * BEFORE the U1 inter-procedural descent expanded it). The callgraph bridge + * keys its "first-hop proven" set on this, NOT the interproc-expanded + * `reachableBlocks` superset, so statementPrecision keeps its first-hop meaning + * (FIX 6). Equals `reachableBlocks` when no hop crossed; empty for the + * no-body / no-block-at-line empty returns. + */ + intraReachableBlocks: string[]; + /** The criterion's own seed blocks (changed statement / whole-symbol body). */ + seedBlocks: string[]; + blockCount: number; + affectedStatements: PdgStatement[]; + affectedStatementCount: number; + depthReached: number; + unresolvedBlockCount: number; + ambiguousProjectionCount: number; + criterionLine?: number; + truncated?: boolean; + truncatedBy?: 'depth' | 'limit'; + truncatedByReasons?: readonly ('depth' | 'limit')[]; +} + +export interface PdgImpactSuccessResult extends PdgImpactBaseResult, PdgImpactSliceFields { + target: Required; + epistemic: 'pdg-intra-procedural'; +} + +export interface PdgImpactEmptyResult extends PdgImpactBaseResult, PdgImpactSliceFields { + target: Required; + epistemic: 'no-pdg-body' | 'pdg-no-block-at-line' | 'pdg-intra-procedural'; +} + +export type PdgDegradedLayerState = Exclude; +export type PdgDegradedLayerStatus = PdgLayerStatus & { state: PdgDegradedLayerState }; + +export interface PdgImpactDegradedResult extends PdgImpactBaseResult { + pdgLayer: PdgDegradedLayerState; + missingSubLayer?: PdgSubLayer; + probeError?: string; + recoverySuggestion?: string; +} + +export interface PdgImpactErrorResult { + mode?: 'pdg'; + /** Contract version of the mode:'pdg' impact result shape; bump on any breaking change to the PDG result fields. */ + pdgResultVersion: 1; + error: string; + target: PdgImpactTarget; + direction: 'upstream' | 'downstream'; + impactedCount: 0; + risk: 'UNKNOWN'; + suggestion?: string; + recoverySuggestion?: string; +} + +export type PdgImpactResult = + | PdgImpactSuccessResult + | PdgImpactEmptyResult + | PdgImpactDegradedResult + | PdgImpactErrorResult; + +export function makePdgImpactErrorResult(input: { + error: string; + target: PdgImpactTarget; + direction: 'upstream' | 'downstream'; + mode?: 'pdg'; + suggestion?: string; + recoverySuggestion?: string; +}): PdgImpactErrorResult { + return { + ...(input.mode ? { mode: input.mode } : {}), + pdgResultVersion: PDG_RESULT_VERSION, + error: input.error, + target: input.target, + direction: input.direction, + impactedCount: 0, + risk: 'UNKNOWN', + ...(input.suggestion ? { suggestion: input.suggestion } : {}), + ...(input.recoverySuggestion ? { recoverySuggestion: input.recoverySuggestion } : {}), + }; +} + +export function isPdgDegradedLayerStatus(layer: PdgLayerStatus): layer is PdgDegradedLayerStatus { + return layer.state !== 'ready'; +} + +export function makePdgLayerDegradedResult(input: { + mode: 'pdg'; + target: PdgImpactTarget; + direction: 'upstream' | 'downstream'; + layer: PdgDegradedLayerStatus; +}): PdgImpactDegradedResult { + return { + mode: input.mode, + pdgResultVersion: PDG_RESULT_VERSION, + pdgLayer: input.layer.state, + ...(input.layer.missingSubLayer ? { missingSubLayer: input.layer.missingSubLayer } : {}), + ...(input.layer.probeError ? { probeError: input.layer.probeError } : {}), + ...(input.layer.recoverySuggestion + ? { recoverySuggestion: input.layer.recoverySuggestion } + : {}), + note: input.layer.note, + target: input.target, + direction: input.direction, + impactedCount: 0, + risk: 'UNKNOWN', + ...emptyPdgParityFields(), + }; +} + +/** + * Assemble the consumer-safe PDG impact result (U4 / KTD8 parity matrix). + * + * Takes the U3 traversal output (reachable block set + truncation signalling) + * plus the U4 block→symbol projection, and shapes a result STRUCTURALLY + * substitutable for the call-graph `_runImpactBFS` result so every consumer + * (CLI `formatImpactResult`, group `collectImpactSymbolUids`/`mergeRisk`, + * `impactByUid`) renders it without misrendering. This is a STANDING + * interchangeability contract, not a one-time check. + * + * Field-by-field vs the call-graph result (KTD8): + * - `target.id/name/type/filePath` — identical shape (`collectImpactSymbolUids` + * keys on `target.id`/`target.filePath`). + * - `byDepth` — same `{ [depth]: item[] }` map shape, but COLLAPSED to a single + * bucket (`1`): intra-procedural dependence has no meaningful inter-symbol hop + * count (block-hops are NOT call-hops). Items carry `{ id, name, type, + * filePath, … }` exactly like the call-graph items so `collectImpactSymbolUids` + * collects their UIDs. `unresolved` shadow-path entries keep `id: null` (they + * are surfaced, never dropped — but collect as no UID). + * - `byDepthCounts` — `{ 1: }`, same shape. + * - `affected_processes` / `affected_modules` — empty `[]` (no + * STEP_IN_PROCESS/module edges originate from BasicBlocks; consumers coalesce + * `[]` safely). + * - `epistemic` — a PDG-specific marker (`'pdg-intra-procedural'`), NOT the + * callgraph DI/dynamic-dispatch `'lower-bound'` copy. `note` carries the + * PDG framing so the CLI prints PDG text, not callgraph boundary text. + * - `risk` — the existing `'UNKNOWN'` sentinel (NOT a new label). `mergeRisk` + * already coalesces `'UNKNOWN'` correctly (never a confident `LOW`). + * - `impactedCount` — count of DISTINCT owning SYMBOLS (resolved UIDs), the + * meaningful unit for the impact question ("which symbols are affected"). + * `blockCount` is retained separately as the raw reachable-block count. + */ +function assemblePdgImpactResult(input: { + target: { id: string; name: string; type: string; filePath: string }; + direction: 'upstream' | 'downstream'; + reachableBlocks: string[]; + /** + * INTRA-procedural reachable subset (before the U1 descent). Surfaced verbatim + * so the callgraph bridge keys its "first-hop proven" set on the original + * statement slice, not the interproc-expanded superset (FIX 6). + */ + intraReachableBlocks: string[]; + /** + * The criterion's own seed blocks (the changed statement / whole-symbol body). + * Surfaced so the dispatcher can prove inter-procedural callees invoked + * directly on the changed line, which are NOT in `reachableBlocks` (the + * seed-minus-reachable convention — seeds are the target, not dependents). + */ + seedBlocks: string[]; + /** Reachable blocks resolved to source statements (the useful slice output). */ + affectedStatements?: PdgStatement[]; + /** The 1-based source line the slice was seeded on (statement mode only). */ + criterionLine?: number; + projection: { symbols: OwningSymbol[]; unresolvedCount: number; ambiguousCount: number }; + depthReached: number; + truncated: boolean; + truncatedBy?: 'depth' | 'limit'; + truncatedByReasons?: readonly ('depth' | 'limit')[]; + /** + * Number of inter-procedural FUNCTION hops the U1 forward closure descended + * (0 ⇒ intra-only, e.g. a pre-namespace-v4 index or a leaf statement). When >0 + * the slice crossed function boundaries, so the note documents the 4 soundness + * caveats of the context-insensitive descent. + */ + interproceduralHops?: number; + /** + * FU-C: whether the index carries CALL_SUMMARY (return-value ascent). When + * `false` AND the slice crossed ≥1 inter-procedural hop, the note flags that a + * PRE-FU-C (v3) index served only the intra slice with no return-value ascent + * and steers to a re-index. `true` ⇒ ascent active (no extra note). + */ + callSummaryAvailable?: boolean; +}): PdgImpactSuccessResult { + const { target, direction, reachableBlocks, projection } = input; + const { symbols, unresolvedCount, ambiguousCount } = projection; + const affectedStatements = input.affectedStatements ?? []; + const statementMode = typeof input.criterionLine === 'number'; + + // Items for the single collapsed bucket. Shaped like the call-graph byDepth + // items (`{ depth, id, name, type, filePath, processes }`) so consumers that + // iterate byDepth read the same fields. `unresolved` entries keep `id: null` + // (surfaced under their file; `collectImpactSymbolUids` skips a null id, which + // is correct — there is no symbol UID to attribute). + const items = symbols.map((s) => ({ + depth: 1, + id: s.id, + name: s.name, + type: s.type, + filePath: s.filePath, + ...(s.startLine !== undefined ? { startLine: s.startLine } : {}), + ...(s.ambiguous ? { ambiguous: true } : {}), + ...(s.id === null ? { unresolved: true } : {}), + pdgEvidence: (s.id === null ? 'degraded' : 'owner-projection') as PdgImpactEvidence, + pdgEvidenceReason: + s.id === null + ? 'reachable BasicBlock has no owning Function/Method/Constructor projection' + : 'reachable BasicBlock projected to its owning symbol', + processes: [] as unknown[], + })); + + // impactedCount = distinct owning SYMBOLS (resolved UIDs). Unresolved shadow + // entries are surfaced in byDepth but do NOT inflate the symbol count. + const resolvedUids = new Set(symbols.filter((s) => s.id !== null).map((s) => s.id as string)); + const impactedCount = resolvedUids.size; + + const byDepth: Record = items.length > 0 ? { 1: items } : {}; + const byDepthCounts: Record = { 1: items.length }; + + const noteParts: string[] = statementMode + ? [ + `mode:'pdg' — intra-procedural slice from line ${input.criterionLine} of ` + + `'${target.name}'. ${affectedStatements.length} ` + + `${affectedStatements.length === 1 ? 'statement is' : 'statements are'} ${direction}-` + + `dependent on it (over CDG + REACHING_DEF). Inter-procedural symbol reach ` + + `is attached by impact mode's unified PDG dispatcher in interproceduralByDepth/byDepth.`, + ] + : [ + `mode:'pdg' — intra-procedural Program Dependence Graph. ${impactedCount} owning ` + + `${impactedCount === 1 ? 'symbol' : 'symbols'} reached via ${reachableBlocks.length} ` + + `dependence ${reachableBlocks.length === 1 ? 'block' : 'blocks'} ` + + `(${direction} over CDG + REACHING_DEF). Inter-procedural symbol reach ` + + `is attached by impact mode's unified PDG dispatcher in interproceduralByDepth/byDepth.`, + ]; + const interproceduralHops = input.interproceduralHops ?? 0; + if (interproceduralHops > 0) { + noteParts.push( + `The statement slice (affectedStatements) crosses ${interproceduralHops} ` + + `inter-procedural ${interproceduralHops === 1 ? 'hop' : 'hops'} via resolved call ` + + `sites (HRB context-insensitive forward closure, downstream). SOUNDNESS CAVEATS: ` + + `(1) context-insensitive — a dependence may be attributed to a callee only reachable ` + + `from a DIFFERENT call site of the same function (bounded over-inclusion, the same ` + + `imprecision callgraph mode has); (2) return-value ascent IS captured (via ` + + `CALL_SUMMARY): a caller statement that depends on a callee's RETURN value is in the ` + + `slice when the callee has a persisted return-flow summary. Out-parameter / mutated-` + + `argument ascent, callee-written shared / captured variables, and exception ascent (a ` + + `throw the caller catches) remain deferred (they need an alias / try-catch model). A ` + + `PRE-FU-C (v3) --pdg index has no CALL_SUMMARY edges, so return-value ascent is absent ` + + `there until a re-index; (3) no cross-boundary alias ` + + `model; (4) precision is bounded by the call RESOLVER's precision (multi-candidate ` + + `dispatch / C++ overload under-resolution flow through faithfully — sound, never drops a ` + + `real target).`, + ); + // FU-C degradation: the slice crossed a call boundary but this index predates + // CALL_SUMMARY (a v3 `--pdg` index). The intra slice is served, but no + // return-value ascent ran — steer to a re-index. + if (input.callSummaryAvailable === false) { + noteParts.push( + `no return-value ascent (re-index for CALL_SUMMARY): this --pdg index predates the ` + + `FU-C return-value-ascent layer, so a caller statement depending on a callee's ` + + `RETURN value is NOT in the slice. Re-run gitnexus analyze --pdg to record ` + + `CALL_SUMMARY edges and enable it.`, + ); + } else if (input.callSummaryAvailable === true) { + // The CALL_SUMMARY layer is present, but return-value ascent is populated + // ONLY for TypeScript/JavaScript today (the formal-index it needs is set + // solely by the TS/JS harvester). For a criterion in any other language the + // ascent is structurally empty, so say so rather than letting the omission + // read as "ascent ran and found nothing". Sound — never claims ascent fired. + // Language is derived HERE in mcp/local, which may name languages; the + // shared core/ingestion pipeline must not. + const lang = getProviderForFile(target.filePath)?.id; + const ascentLanguage = + lang === SupportedLanguages.TypeScript || lang === SupportedLanguages.JavaScript; + if (!ascentLanguage) { + noteParts.push( + `return-value ascent is currently TypeScript/JavaScript-only (only the TS/JS harvester ` + + `records the formal-index it needs), so a caller statement depending on a non-TS/JS ` + + `callee's RETURN value is not in the slice. Descent and the intra slice are unaffected.`, + ); + } + } + } + if (ambiguousCount > 0) { + noteParts.push( + `${ambiguousCount} owning-symbol ${ambiguousCount === 1 ? 'projection is' : 'projections are'} ` + + `ambiguous: same-line functions cannot be disambiguated by start line alone (no startColumn ` + + `in the schema), so ALL colliding symbols are reported — none is silently picked.`, + ); + } + if (unresolvedCount > 0) { + noteParts.push( + `${unresolvedCount} reachable ${unresolvedCount === 1 ? 'block maps' : 'blocks map'} to no ` + + `owning Function/Method/Constructor (top-level statement or a lambda whose start line is not a symbol ` + + `start) — surfaced under their file as 'unresolved', never dropped.`, + ); + } + + return { + mode: 'pdg', + pdgResultVersion: PDG_RESULT_VERSION, + target, + direction, + impactedCount, + // KTD8: reuse the existing UNKNOWN sentinel — never a confident LOW (which + // would read as "safe to refactor"; #2129/#1858 false-safe lineage). PDG + // mode is intra-procedural, so its count is a per-function lower bound on the + // true blast radius and risk is genuinely UNKNOWN at the program level. + risk: 'UNKNOWN', + // PDG-specific epistemic marker — NOT the callgraph 'lower-bound'/DI copy. + epistemic: 'pdg-intra-procedural', + note: noteParts.join(' '), + pdgEvidence: { + statements: 'local-dependence', + localSymbols: unresolvedCount > 0 ? 'degraded' : 'owner-projection', + localSymbolCount: impactedCount, + unresolvedBlockCount: unresolvedCount, + ambiguousProjectionCount: ambiguousCount, + }, + // Statement-level slice: the dependent source statements (line + text) the + // change reaches. This is the primary useful output of statement mode; the + // accuracy harness scores against these lines. + ...(statementMode ? { criterionLine: input.criterionLine } : {}), + affectedStatements, + affectedStatementCount: affectedStatements.length, + // Raw block-level detail retained alongside the symbol projection (U3 tests + // and the accuracy harness read these). + reachableBlocks, + intraReachableBlocks: input.intraReachableBlocks, + seedBlocks: input.seedBlocks, + blockCount: reachableBlocks.length, + depthReached: input.depthReached, + unresolvedBlockCount: unresolvedCount, + ambiguousProjectionCount: ambiguousCount, + ...(input.truncated ? { truncated: true } : {}), + ...(input.truncatedBy ? { truncatedBy: input.truncatedBy } : {}), + ...(input.truncatedByReasons ? { truncatedByReasons: input.truncatedByReasons } : {}), + summary: { + direct: impactedCount, + processes_affected: 0, + modules_affected: 0, + }, + byDepthCounts, + affected_processes: [] as unknown[], + affected_modules: [] as unknown[], + byDepth, + }; +} + +/** The two impact engines (KTD1). `'callgraph'` is the default/established path. */ +export type ImpactMode = 'callgraph' | 'pdg'; + +/** + * Validate the `impact` `mode` param (KTD5 — backend hard-gate). + * + * The MCP JSON-schema `enum` is advisory only (server.ts forwards args + * unvalidated and `callTool` is reachable directly), so this backend check is + * the real boundary — mirroring `_pdgQueryImpl`'s `mode` enum validation. A + * typo'd mode silently running callgraph is exactly the silent fallback this + * forbids (it would make the accuracy harness compare callgraph-vs-callgraph + * and report perfect parity). + * + * Absent / `undefined` / `'callgraph'` all resolve to `'callgraph'` (the + * unchanged default path). `'pdg'` is valid. Anything else — `'PDG'`, `'pgd'`, + * `''`, or a non-string (`0`, `null`, …) — returns a structured `{ error }`, + * never a callgraph result. + */ +export function validateImpactMode(rawMode: unknown): { mode: ImpactMode } | { error: string } { + if (rawMode === undefined || rawMode === 'callgraph') return { mode: 'callgraph' }; + if (rawMode === 'pdg') return { mode: 'pdg' }; + return { + error: `Invalid "mode": expected "callgraph" or "pdg", got ${JSON.stringify(rawMode)}.`, + }; +} + +/** The two independently-stamped PDG sub-layers (KTD7). */ +export type PdgSubLayer = 'CDG' | 'REACHING_DEF'; + +/** + * Four-state PDG-layer presence/degradation status (KTD7). + * + * - `'no-layer'` — `meta.pdg` is absent: this repo was never analyzed + * with `--pdg` (definitive; established with NO DB scan). + * - `'sub-layer-missing'`— exactly one of the two independently-stamped caps + * (`maxCdgEdgesPerFunction` / `maxReachingDefEdgesPerFunction`) + * is present. `impact`'s PDG mode needs BOTH, so a + * partial layer must not be reported as complete; the + * missing one is named in `missingSubLayer`. + * - `'ready'` — both caps present: the layer is fully stamped. + * - `'unknown'` — meta is unreadable (e.g. a seeded test DB with no + * `meta.json`). One bounded `LIMIT 1` probe distinguishes + * a genuinely edge-free index from a missing one; either + * way the conclusion is inconclusive (a missing layer is + * indistinguishable from an all-linear one — #2188). + */ +export interface PdgLayerStatus { + state: 'no-layer' | 'sub-layer-missing' | 'ready' | 'unknown'; + /** Set only for `'sub-layer-missing'` — the cap that was NOT stamped. */ + missingSubLayer?: PdgSubLayer; + /** Human-readable guidance for the degraded states (absent for `'ready'`). */ + note?: string; + /** Set when an unknown-state probe failed before it could inspect PDG rows. */ + probeError?: string; + /** Optional operator-facing recovery hint for probe failures. */ + recoverySuggestion?: string; + /** + * FU-C return-value-ascent layer presence (read from `meta.pdg.hasCallSummary`). + * `true` ⇒ CALL_SUMMARY edges exist, so return-value ascent is active. `false` + * ⇒ a PRE-FU-C (v3) `--pdg` index — the intra slice is unaffected, but the + * result should NOTE "no return-value ascent (re-index for CALL_SUMMARY)". + * Meaningful only for the meta-readable states (`'ready'` / `'sub-layer-missing'` + * / `'no-layer'`); `undefined` for `'unknown'` (meta unreadable). Deliberately + * NOT part of the `'ready'` gate — a v3 index is still `'ready'` for the slice. + */ + hasCallSummary?: boolean; +} + +/** + * Per-cap presence read from `meta.pdg`, plus whether meta was readable at all. + * + * `metaReadable` is the seam between the `'unknown'` state (meta unreadable — + * fall through to a DB probe) and the meta-stamped states. When `metaReadable` + * is true but `meta.pdg` was absent, both `cdg`/`rd` are `false`. + */ +interface PdgMetaCaps { + metaReadable: boolean; + /** `maxCdgEdgesPerFunction !== undefined` (only meaningful when metaReadable). */ + cdg: boolean; + /** `maxReachingDefEdgesPerFunction !== undefined` (only meaningful when metaReadable). */ + rd: boolean; + /** + * `pdg.hasCallSummary === true` (only meaningful when metaReadable). The FU-C + * return-value-ascent layer is OPTIONAL — its absence does NOT degrade + * `pdgLayerStatus` below `'ready'` (a v3 index still serves the intra slice); + * it only suppresses (and flags) the ascent. Read here so the impact note can + * steer a pre-FU-C index to re-index without a separate meta probe. + */ + callSummary: boolean; +} + +/** + * Read the two PDG sub-layer caps from the on-disk `meta.json` stamp — the + * single shared meta-probe both `_pdgQueryImpl` (one cap) and the PDG impact + * mode (both caps) key on. Never scans the DB. An unreadable / missing meta + * yields `metaReadable: false` (the `'unknown'` seam); a readable meta with no + * `pdg` stamp yields `metaReadable: true` with both caps `false` (no-layer). + */ +async function readPdgMetaCaps( + lbugPath: string, + loadMetaFn: typeof loadMeta, +): Promise { + try { + const meta = await loadMetaFn(path.dirname(lbugPath)); + if (!meta) return { metaReadable: false, cdg: false, rd: false, callSummary: false }; + return { + metaReadable: true, + cdg: meta.pdg?.maxCdgEdgesPerFunction !== undefined, + rd: meta.pdg?.maxReachingDefEdgesPerFunction !== undefined, + callSummary: meta.pdg?.hasCallSummary === true, + }; + } catch { + // Meta unreadable — the caller decides from the DB (the `'unknown'` state). + return { metaReadable: false, cdg: false, rd: false, callSummary: false }; + } +} + +/** + * Project the both-caps PDG meta read down to the single mode-relevant cap that + * `_pdgQueryImpl` keys on (`controls` → CDG, `flows` → REACHING_DEF), preserving + * its established tri-state `boolean | undefined` contract byte-for-byte + * (Feasibility Issue 4): + * - `false` — meta readable and the relevant cap absent → definitive + * no-layer (short-circuits before any DB scan). + * - `true` — meta readable and the relevant cap present → proceed. + * - `undefined` — meta unreadable → defer to the post-anchored-query probe. + * + * `_pdgQueryImpl` needs only ONE cap, so it collapses the both-caps read here + * rather than consuming `pdgLayerStatus` directly (whose `'unknown'` state does + * an upfront global probe — wrong timing/order for the anchored-query path). + */ +export async function pdgStampForMode( + lbugPath: string, + mode: 'controls' | 'flows', + loadMetaFn: typeof loadMeta = loadMeta, +): Promise { + const caps = await readPdgMetaCaps(lbugPath, loadMetaFn); + if (!caps.metaReadable) return undefined; + return mode === 'controls' ? caps.cdg : caps.rd; +} + +/** + * PDG-layer presence/degradation check for the `impact` PDG mode (KTD7). + * + * Returns the four distinct states WITHOUT scanning the DB except for the single + * bounded `LIMIT 1` probe the `'unknown'` (meta-unreadable) case requires. The + * caller (`_impactImpl` PDG branch, and the accuracy harness) surfaces a + * distinct signal per state so a missing `--pdg` layer / partial layer is never + * silently misread as a confident empty blast radius. Impact needs BOTH the CDG + * and the REACHING_DEF sub-layer, so a partial stamp degrades, not proceeds. + */ +export async function pdgLayerStatus(deps: { + lbugPath: string; + executeParameterized: typeof executeParameterized; + loadMetaFn?: typeof loadMeta; +}): Promise { + const loadMetaFn = deps.loadMetaFn ?? loadMeta; + const caps = await readPdgMetaCaps(deps.lbugPath, loadMetaFn); + + if (caps.metaReadable) { + // Meta is readable — the stamp is authoritative, no DB scan needed. + // `hasCallSummary` rides on every meta-readable state (it is NOT part of the + // `'ready'` gate — a v3 index without it is still ready for the intra slice). + if (caps.cdg && caps.rd) return { state: 'ready', hasCallSummary: caps.callSummary }; + if (caps.cdg !== caps.rd) { + // Exactly one sub-layer stamped (XOR) — partial layer; impact needs both. + const missingSubLayer: PdgSubLayer = caps.cdg ? 'REACHING_DEF' : 'CDG'; + return { + state: 'sub-layer-missing', + missingSubLayer, + hasCallSummary: caps.callSummary, + note: + `PDG layer is incomplete — the ${missingSubLayer} sub-layer is missing ` + + `(impact's PDG mode needs both CDG and REACHING_DEF). ` + + `Re-run gitnexus analyze --pdg to record it.`, + }; + } + // Neither cap stamped (meta.pdg absent, or present with no caps) → the layer + // was never recorded. Definitive, no DB scan. + return { + state: 'no-layer', + hasCallSummary: caps.callSummary, + note: 'no PDG layer — run gitnexus analyze --pdg to record CDG + REACHING_DEF edges for this repo', + }; + } + + // Meta unreadable (e.g. a seeded test DB): one bounded probe confirms the + // layer status is genuinely undeterminable from the DB. A missing layer is + // indistinguishable from an all-linear (edge-free) one (#2188), so whether the + // probe finds a row or not the state stays `'unknown'` (never the definitive + // no-layer wording). The probe is bounded (`LIMIT 1`) and anchored on the + // BasicBlock→BasicBlock partition (the `(:BasicBlock)…(:BasicBlock)` label pair + // restricts it to the sparse pdg-edge partition, never a global rel scan — the + // established `_explainImpl` anchoring pattern), and it is wrapped so a db-lock + // / missing-path throw degrades to the same `'unknown'` signal rather than + // propagating and losing it. + // + // The probe result is NOT discarded: a visible CDG/REACHING_DEF edge (with + // meta unreadable) is a weak-but-real "edges are present, but completeness is + // unprovable" signal, distinct from "no edges visible at all". Both stay + // `'unknown'` (inconclusive), but the note distinguishes them so the operator + // gets the more useful hint. + let edgesVisible = false; + let probeError: string | undefined; + try { + const rows = await deps.executeParameterized( + deps.lbugPath, + `MATCH (:BasicBlock)-[r:CodeRelation]->(:BasicBlock) WHERE r.type IN ['CDG', 'REACHING_DEF'] RETURN r.type AS type LIMIT 1`, + {}, + ); + edgesVisible = Array.isArray(rows) && rows.length > 0; + } catch (err) { + // db-lock / missing-path / corrupt probe — fall through as not-visible, but + // keep the `'unknown'` signal AND preserve the probe failure. Reporting the + // failed probe as "no edges visible" hides a DB-health problem from operators. + probeError = err instanceof Error ? err.message : String(err); + edgesVisible = false; + } + if (probeError) { + return { + state: 'unknown', + probeError, + recoverySuggestion: + 'Check for a LadybugDB lock/corruption or missing index path. Stop overlapping GitNexus processes, retry, or re-run gitnexus analyze --pdg.', + note: + `PDG layer status unknown — CDG/REACHING_DEF probe failed: ${probeError}. ` + + `The layer cannot be confirmed complete; this is distinct from "no edges visible".`, + }; + } + return { + state: 'unknown', + note: edgesVisible + ? 'PDG layer status unknown — CDG/REACHING_DEF edges ARE visible but meta is unreadable, so the layer cannot be confirmed complete (a partial layer looks the same); was this repo fully indexed with gitnexus analyze --pdg?' + : 'PDG layer status unknown — no CDG/REACHING_DEF edges visible and meta is unreadable; was this repo indexed with gitnexus analyze --pdg?', + }; +} + +/** + * Build the SAME BasicBlock seed anchor (`anchorClause` + `queryParams`) as + * `resolveBlockAnchor`'s symbol branch, but from an ALREADY-RESOLVED symbol — + * WITHOUT re-running `resolveSymbolCandidates`. + * + * Why this exists (correctness keystone): `_impactImpl` already resolves the + * target to a confident single symbol honoring the caller's + * `target_uid`/`file_path`/`kind` hints. Re-resolving by the bare `sym.name` + * inside `_runImpactPDG` would (a) RE-AMBIGUATE a globally-ambiguous name the + * caller had disambiguated (returning the "ambiguous" early payload instead of + * the PDG result), or (b) anchor the seed on a DIFFERENT same-name symbol in + * another file → a wrong-symbol blast radius. Anchoring directly from the + * resolved `{ filePath, startLine, endLine }` preserves the disambiguation. + * + * The window is byte-identical to `resolveBlockAnchor`'s symbol branch: BOTH + * span bounds are shifted `+1` (1-based BasicBlock `startLine` vs the 0-based + * symbol span — the lower `+1` excludes a neighbor's block on the line above, + * the upper `+1` keeps a guard/def/use on the final line). A symbol with no + * usable span degrades to the same file-level id-prefix filter. This is the + * resolved-symbol counterpart, NOT a second window convention. + */ +function blockAnchorForResolvedSymbol(sym: { + filePath: string; + startLine?: number; + endLine?: number; +}): { anchorClause: string; queryParams: Record } { + const idPrefix = `BasicBlock:${sym.filePath}:`; + if ( + typeof sym.startLine === 'number' && + typeof sym.endLine === 'number' && + sym.endLine >= sym.startLine + ) { + return { + anchorClause: + 'a.id STARTS WITH $idPrefix AND a.startLine >= $symStart AND a.startLine <= $symEnd', + queryParams: { idPrefix, symStart: sym.startLine + 1, symEnd: sym.endLine + 1 }, + }; + } + return { anchorClause: 'a.id STARTS WITH $idPrefix', queryParams: { idPrefix } }; +} + +/** + * Build a STATEMENT seed anchor: the BasicBlock(s) starting at a specific + * 1-based source `line` WITHIN the resolved symbol. This is what makes + * `mode:'pdg'` useful — seeding the dependence slice on a single statement + * (the thing being changed) rather than the whole symbol. A whole-symbol seed + * captures every intra-procedural block, so the reachable-minus-seed set is + * empty (all intra reach is within the seed); a statement seed leaves the + * other dependent statements reachable. `BasicBlock.startLine` is 1-based and + * matches the source line, so no `+1` offset applies here (unlike the symbol + * span, where the 0-based symbol bounds are shifted). Bounded to the symbol's + * own span when known, so a line shared with a sibling symbol can't leak. + */ +function blockAnchorForStatement( + sym: { filePath: string; startLine?: number; endLine?: number }, + line: number, +): { anchorClause: string; queryParams: Record } { + const idPrefix = `BasicBlock:${sym.filePath}:`; + if ( + typeof sym.startLine === 'number' && + typeof sym.endLine === 'number' && + sym.endLine >= sym.startLine + ) { + return { + anchorClause: + 'a.id STARTS WITH $idPrefix AND a.startLine = $line AND a.startLine >= $symStart AND a.startLine <= $symEnd', + queryParams: { idPrefix, line, symStart: sym.startLine + 1, symEnd: sym.endLine + 1 }, + }; + } + return { + anchorClause: 'a.id STARTS WITH $idPrefix AND a.startLine = $line', + queryParams: { idPrefix, line }, + }; +} + +/** + * One bounded, direction-aware BFS over CDG + REACHING_DEF starting from a set + * of seed blocks. Extracted (U1) from `runImpactPDG`'s intra loop so the + * inter-procedural descent reuses the EXACT same edge query and step/limit + * semantics — no reimplementation. The query, the one-past-`stepLimit` probe, + * the per-step truncation flag, and the depth-exhaustion flag are byte-identical + * to the original inline loop. + * + * `visited` is the caller's shared cycle/recursion guard: seeds are pre-added so + * they are never re-collected (the seed-minus-reachable convention), and any + * block already visited (a prior callee's seed, or already-reached) is skipped. + * Newly-discovered blocks are added to `visited` AND returned in `reachable`. + */ +async function bfsReachableBlocks(input: { + lbugPath: string; + exec: typeof executeParameterized; + seedBlocks: string[]; + visited: Set; + direction: 'upstream' | 'downstream'; + depthBudget: number; + stepLimit: number; + probeLimit: number; +}): Promise<{ + reachable: Set; + depthReached: number; + truncatedByDepth: boolean; + truncatedByLimit: boolean; +}> { + const { lbugPath, exec, seedBlocks, visited, direction, depthBudget, stepLimit, probeLimit } = + input; + const reachable = new Set(); + // Seeds are pre-added to `visited` so they are never re-COLLECTED (the + // seed-minus-reachable convention), but the seed list still drives the first + // step's frontier — the guard blocks re-collection, not seed re-expansion. + for (const id of seedBlocks) visited.add(id); + let frontier = [...seedBlocks]; + let depthReached = 0; + let truncatedByDepth = false; + let truncatedByLimit = false; + + const matchEndpoint = direction === 'downstream' ? 'a' : 'b'; + const collectEndpoint = direction === 'downstream' ? 'b' : 'a'; + + for (let depth = 0; depth < depthBudget; depth++) { + if (frontier.length === 0) break; + const rawRows = await exec( + lbugPath, + `MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock) + WHERE r.type IN ['CDG', 'REACHING_DEF'] AND ${matchEndpoint}.id IN $frontier + RETURN DISTINCT ${collectEndpoint}.id AS id + LIMIT ${probeLimit}`, + { frontier }, + ); + // Narrow the awaited rows ONCE at the boundary (executeParameterized returns + // any[]) to a typed record shape, then read the aliased `id` via bracket + // access — no `as any` sprayed per field. + const rows = rawRows.slice(0, stepLimit) as Array>; + depthReached = depth + 1; + if (rawRows.length > stepLimit) truncatedByLimit = true; + + const next: string[] = []; + for (const r of rows) { + const id = String(r['id'] ?? ''); + if (!id || visited.has(id)) continue; + visited.add(id); + reachable.add(id); + next.push(id); + } + frontier = next; + } + if (frontier.length > 0) truncatedByDepth = true; + + return { reachable, depthReached, truncatedByDepth, truncatedByLimit }; +} + +/** A resolved callee symbol span — the seed window for an inter-procedural hop. */ +interface CalleeSpan { + id: string; + filePath: string; + startLine: number; + endLine: number; +} + +/** + * Gather the resolved callee symbol ids invoked across a set of slice blocks + * (`BasicBlock.calleeIds`). Reuses the SHARED `splitCalleeIds` so the descent + * cannot diverge from `LocalBackend.calleeIdsOfBlocks` on the split/drop-sentinel + * logic. A pre-namespace-v4 index (no `calleeIds` column → empty cells) yields no + * ids, so the descent degrades cleanly to intra-only (no inter-procedural hop). + */ +async function calleeIdsFromBlocks( + lbugPath: string, + blockIds: string[], + exec: typeof executeParameterized, +): Promise> { + const ids = new Set(); + for (const { calleeIds } of await calleeIdsByBlock(lbugPath, blockIds, exec)) { + for (const id of calleeIds) ids.add(id); + } + return ids; +} + +/** One slice block paired with the resolved callee ids it invokes. */ +interface BlockCallees { + blockId: string; + calleeIds: string[]; +} + +/** + * Per-block variant of {@link calleeIdsFromBlocks}: keep the CALL block → its + * `calleeIds` mapping rather than flattening it. The return-value ascent (U-C4) + * needs this association — it re-seeds the caller's intra closure FROM the + * specific call block whose callee's `CALL_SUMMARY` licenses the ascent, so the + * flattened id-only set is insufficient. Reuses the SHARED `splitCalleeIds` so + * the split/drop-sentinel logic cannot diverge from the flattening caller. A + * block with no callee ids (empty/whitespace cell, or a pre-v4 index with no + * `calleeIds` column) yields an empty `calleeIds` — skipped by the consumer. + */ +async function calleeIdsByBlock( + lbugPath: string, + blockIds: string[], + exec: typeof executeParameterized, +): Promise { + if (blockIds.length === 0) return []; + const rows = await exec( + lbugPath, + `MATCH (b:BasicBlock) WHERE b.id IN $ids RETURN b.id AS id, b.calleeIds AS calleeIds`, + { ids: blockIds }, + ); + const out: BlockCallees[] = []; + // Narrow the awaited rows ONCE at the boundary to a typed record shape; read + // the aliased cells via bracket access — no per-field `as any`. + for (const r of rows as Array>) { + const blockId = String(r['id'] ?? ''); + if (!blockId) continue; + const calleeIds = splitCalleeIds(r['calleeIds']); + if (calleeIds.length > 0) out.push({ blockId, calleeIds }); + } + return out; +} + +/** + * Of a set of resolved callee symbol ids, which ones have a persisted + * `CALL_SUMMARY` self-loop edge recording a NON-EMPTY return-value ascent + * (≥1 formal parameter flows to the callee's return). This is the FU-C consumer + * side of the producer's per-callee summary (see `call-summary-codec.ts`). + * + * The summary is a self-loop on the Function/Method/Constructor node: + * `(c)-[r:CodeRelation {type:'CALL_SUMMARY'}]->(c) WHERE c.id IN $ids`. The + * `reason` carries the param→return bitset; `decodeCallSummary` unpacks it and + * NEVER throws — a malformed / absent / empty (`r:0`) summary yields NO entry + * (the sound default: never claim a false return-flow). A PRE-FU-C (v3) `--pdg` + * index has NO `CALL_SUMMARY` edges, so this returns the empty set and the + * ascent is a clean no-op (the intra slice is unchanged — the documented + * "re-index for CALL_SUMMARY" degradation). + */ +async function calleesWithReturnFlow( + lbugPath: string, + calleeIds: string[], + exec: typeof executeParameterized, +): Promise> { + const out = new Set(); + if (calleeIds.length === 0) return out; + const rows = await exec( + lbugPath, + `MATCH (c)-[r:CodeRelation]->(c) + WHERE r.type = 'CALL_SUMMARY' AND c.id IN $ids + RETURN c.id AS id, r.reason AS reason`, + { ids: calleeIds }, + ); + for (const r of rows as Array>) { + const id = String(r['id'] ?? ''); + if (!id) continue; + const decoded = decodeCallSummary(r['reason']); + // ARG→FORMAL trace precision: the conservative-but-sound default — ascend if + // ANY formal is return-flowing (the call site's argument is, by construction + // of the descent, in the slice: the call block is itself a slice block). A + // specific positional arg↔formal mapping is not cleanly recoverable at the + // coalesced call block (`BasicBlock.calleeIds` is an unordered set, not a + // per-arg list), so this never drops a real ascent; it may over-include + // (bounded — the result still flows to a slice statement). See the descent + // doc-comment + the result `note` caveat. + if (decoded.ok && decoded.returnFlowParams.length > 0) out.add(id); + } + return out; +} + +/** + * Batch-resolve resolved callee symbol ids → their `{id,filePath,startLine,endLine}` + * spans via ONE `s.id IN $ids` UNION-ALL query over Function/Method/Constructor — + * the SAME query shape as `projectBlocksToSymbols`, but keyed on the RESOLVED + * `s.id` (no same-line ambiguity, unlike the reverse block-to-symbol join). Ids + * with no matching symbol (a callee resolved to a node kind without a CFG body, + * or an out-of-repo id) simply produce no span and are skipped — never an error. + */ +async function resolveCalleeSpans( + lbugPath: string, + calleeIds: string[], + exec: typeof executeParameterized, +): Promise { + if (calleeIds.length === 0) return []; + const rows = await exec( + lbugPath, + `MATCH (s:\`Function\`) WHERE s.id IN $ids + RETURN s.id AS id, s.filePath AS filePath, s.startLine AS startLine, s.endLine AS endLine + UNION ALL + MATCH (s:\`Method\`) WHERE s.id IN $ids + RETURN s.id AS id, s.filePath AS filePath, s.startLine AS startLine, s.endLine AS endLine + UNION ALL + MATCH (s:\`Constructor\`) WHERE s.id IN $ids + RETURN s.id AS id, s.filePath AS filePath, s.startLine AS startLine, s.endLine AS endLine`, + { ids: calleeIds }, + ); + const spans: CalleeSpan[] = []; + // Narrow the awaited rows ONCE at the boundary to a typed record shape; read + // the aliased columns via bracket access with Number()/String() coercion — + // no per-field `as any` (the same boundary-narrowing the typed helpers use). + for (const r of rows as Array>) { + const id = String(r['id'] ?? ''); + const filePath = String(r['filePath'] ?? ''); + const startLine = Number(r['startLine']); + const endLine = Number(r['endLine']); + if (!id || !filePath || !Number.isFinite(startLine) || !Number.isFinite(endLine)) continue; + spans.push({ id, filePath, startLine, endLine }); + } + return spans; +} + +/** + * Bounded inter-procedural forward closure (U1) — HRB context-INSENSITIVE + * forward slicing, the shipped Joern approach. Starting from the intra-procedural + * slice blocks, descend DOWNSTREAM through resolved call sites: per hop, gather + * the slice blocks' `calleeIds`, resolve them to callee spans, seed each callee's + * blocks, run the SAME CDG+REACHING_DEF BFS within the callee, and union the + * newly-reachable blocks into the slice. Recurses to `depthBudget` FUNCTION hops. + * + * Termination is guaranteed by the shared `visited` set (a block is expanded at + * most once across all hops) plus the depth cap. A total node cap (`nodeBudget`) + * is the secondary guard against a pathological fan-out within the budget. + * + * Caveats (documented in the result note + bench/impact-pdg/README.md): + * (1) context-insensitive — a dependence may be attributed to a callee only + * reachable from a DIFFERENT call site of the same function (bounded + * over-inclusion, the same imprecision callgraph mode already has); + * (2) return-value ascent IS captured (via CALL_SUMMARY, U-C4): a caller + * statement that depends on a callee's RETURN value is re-seeded into the + * caller's continuation when the callee carries a persisted return-flow + * summary. Out-parameter / mutated-argument ascent, callee-written shared / + * captured variables, and exception ascent (a throw the caller catches) + * remain deferred (they need an alias / try-catch model). A pre-FU-C (v3) + * --pdg index has no CALL_SUMMARY edges → no return-value ascent there; + * (3) no cross-boundary alias model; + * (4) precision is bounded by the call RESOLVER's precision (multi-candidate + * dispatch / C++ overload under-resolution flow through faithfully — sound, + * never drops a real target). + */ +async function interproceduralDescent(input: { + lbugPath: string; + exec: typeof executeParameterized; + /** The intra-procedural slice = seed blocks ∪ intra-reachable blocks. */ + initialSliceBlocks: string[]; + /** Shared cycle/recursion guard (already contains the intra seeds + reach). */ + visited: Set; + /** FUNCTION-hop budget — how many call boundaries the closure descends. */ + depthBudget: number; + /** + * Per-callee intra BFS DEPTH budget (dependence-levels within one callee). The + * SAME `Math.min(maxDepth, IMPACT_MAX_DEPTH)` clamp the top-level intra BFS + * applies — NOT `stepLimit` (a row/probe budget): a callee must not be + * traversable up to `PDG_QUERY_MAX_LIMIT` dependence-levels deep, one + * sequential DB query per level. + */ + intraDepthBudget: number; + stepLimit: number; + probeLimit: number; + /** Total newly-reached-block cap across all hops (secondary guard). */ + nodeBudget: number; + /** + * Callee symbol ids to treat as ALREADY seeded before the first hop — chiefly + * the statement seed's OWNER FUNCTION, so direct/mutual recursion back to it + * never re-seeds the whole-function span the statement slice deliberately + * excluded (the statement-precision pin would otherwise be re-broadened). + */ + preSeededCalleeIds?: Iterable; +}): Promise<{ + reachable: Set; + hopsReached: number; + truncatedByDepth: boolean; + truncatedByLimit: boolean; + truncatedByNodeCap: boolean; + /** + * CALL blocks whose callee's `CALL_SUMMARY` licensed a RETURN-VALUE ascent + * (U-C4): the call's RESULT depends on the slice (the call block is a slice + * block), so the caller's intra closure continues THROUGH the result. The + * caller surfaces these blocks' interior source lines (a coalesced call block + * spans the call statements whose results chain through it) — the + * statement-granularity realisation of the ascent. The downstream re-seed from + * the block is already unioned into `reachable`; this set additionally records + * WHICH blocks got the ascent so the statement projection can expand them. + */ + ascentBlocks: Set; +}> { + const { + lbugPath, + exec, + visited, + depthBudget, + intraDepthBudget, + stepLimit, + probeLimit, + nodeBudget, + } = input; + const reachable = new Set(); + let truncatedByDepth = false; + let truncatedByLimit = false; + let truncatedByNodeCap = false; + let hopsReached = 0; + // The slice the next hop gathers callees FROM — starts as the intra slice, then + // becomes each hop's newly-reached callee blocks. + let sliceBlocks = [...input.initialSliceBlocks]; + // Guard against re-seeding the same callee function across hops. Pre-seeded + // with the statement seed's owner function so recursion never re-broadens it. + const seededCalleeIds = new Set(input.preSeededCalleeIds ?? []); + // U-C4 return-value ascent: CALL blocks whose callee has a non-empty + // CALL_SUMMARY return-flow → the call's result depends on the slice. + const ascentBlocks = new Set(); + + hopLoop: for (let hop = 0; hop < depthBudget; hop++) { + if (sliceBlocks.length === 0) break; + // Keep the CALL block → callee association (U-C4 needs it to re-seed the + // caller's intra closure FROM the specific call block the ascent licenses); + // the flattened id set still drives the descent's fresh-callee bookkeeping. + const blockCallees = await calleeIdsByBlock(lbugPath, sliceBlocks, exec); + const calleeIds = new Set(); + for (const { calleeIds: ids } of blockCallees) for (const id of ids) calleeIds.add(id); + + // ── U-C4 ascent: re-seed the caller's intra closure THROUGH call results ── + // For each call block in THIS hop's slice, if ANY of its callees' return + // values flow back (a non-empty CALL_SUMMARY), the call result is + // slice-dependent (the call block is itself a slice block). Re-seed a + // BOUNDED downstream BFS FROM the call block so the caller's continuation + // that consumes the result is captured. Monotone: only ADDS to `reachable`, + // reusing the shared `visited` set, so it stays bounded + terminating. A + // pre-v4 index (no CALL_SUMMARY) yields no return-flowing callees → no-op. + const returnFlowing = await calleesWithReturnFlow(lbugPath, [...calleeIds], exec); + if (returnFlowing.size > 0) { + for (const { blockId, calleeIds: ids } of blockCallees) { + // Bound the ascent re-seeds the same way the descent bounds its per-span + // BFS (line ~1496): a wide fan-out of return-flowing call blocks must not + // run unbounded re-seeds inside a single hop. Mirror the descent's + // node-cap short-circuit at the TOP of the loop, stamping the same flag. + if (reachable.size > nodeBudget) { + truncatedByNodeCap = true; + break hopLoop; + } + if (!ids.some((id) => returnFlowing.has(id))) continue; + // The call block is in the slice by construction; record it so the + // statement projection surfaces the call statements whose results chain + // through it (a coalesced call block spans several call lines). + ascentBlocks.add(blockId); + // Continue the caller's intra closure from the call block. The block is + // already in `visited` (it is a slice block), so the shared-visited BFS + // only adds genuinely-new downstream caller blocks (e.g. a SEPARATE + // statement that uses the call result), never re-expanding the seed. + const ascent = await bfsReachableBlocks({ + lbugPath, + exec, + seedBlocks: [blockId], + visited, + direction: 'downstream', + depthBudget: intraDepthBudget, + stepLimit, + probeLimit, + }); + if (ascent.truncatedByLimit) truncatedByLimit = true; + for (const id of ascent.reachable) reachable.add(id); + } + } + + const freshIds = [...calleeIds].filter((id) => !seededCalleeIds.has(id)); + if (freshIds.length === 0) break; + for (const id of freshIds) seededCalleeIds.add(id); + + const spans = await resolveCalleeSpans(lbugPath, freshIds, exec); + if (spans.length === 0) break; + hopsReached = hop + 1; + + // U13: fetch every callee's seed blocks CONCURRENTLY. The seed fetch was the + // only per-span round-trip; running them in one wave collapses N sequential + // round-trips. Each span runs the IDENTICAL query as before — same anchor, + // same LIMIT, same slice — so the per-span seed set and the `exceeded` + // (truncatedByLimit) flag are byte-identical; only the latency changes. The + // flag is still APPLIED per span in the sequential loop below, so a span past + // the node-budget short-circuit never sets it (prior semantics preserved). + const spanSeeds = await Promise.all( + spans.map(async (span) => { + const { anchorClause, queryParams } = blockAnchorForResolvedSymbol(span); + const rawSeedRows = await exec( + lbugPath, + `MATCH (a:BasicBlock) WHERE ${anchorClause} RETURN a.id AS id LIMIT ${probeLimit}`, + queryParams, + ); + const exceeded = rawSeedRows.length > stepLimit; + const seeds = rawSeedRows + .slice(0, stepLimit) + .map((r: Record) => String(r['id'] ?? '')) + .filter((id: string) => id.length > 0); + return { exceeded, seeds }; + }), + ); + + // U14: run each callee's BFS CONCURRENTLY against a PRIVATE clone of the + // hop-start `visited` snapshot, then merge IN SPAN ORDER below. Cross-span + // sharing of the mutable `visited` set was the ONLY thing forcing the BFS + // sequential; the clones remove the race. The reachable set is the monotone + // union of the per-callee closures (order-independent — `bfsReachableBlocks` + // keys its query on the frontier, not `visited`; `visited` only gates + // re-collection), so only the per-callee BFS DB round-trips run in parallel + // while the merged result stays byte-identical. In a degraded case where a + // per-callee BFS hits its depth/step limit, a sibling may expand through the + // truncated region — a bounded, sound OVER-approximation, never fewer blocks. + const visitedSnapshot = new Set(visited); + const spanBfs = await Promise.all( + spanSeeds.map(async ({ seeds }) => + seeds.length === 0 + ? null + : bfsReachableBlocks({ + lbugPath, + exec, + seedBlocks: seeds, + visited: new Set(visitedSnapshot), // private clone — no cross-span race + direction: 'downstream', + // Per-callee intra DEPTH clamp (block-hops within the callee), NOT + // the row budget — mirrors the top-level intra BFS depth clamp. + depthBudget: intraDepthBudget, + stepLimit, + probeLimit, + }), + ), + ); + + const hopReached = new Set(); + for (let si = 0; si < spans.length; si++) { + // Node budget is checked INSIDE the per-span MERGE (in span order) so the + // mid-hop short-circuit stays byte-identical: the cumulative reachable size + // after merging spans 0..k equals the sequential path's, so the break fires + // at the SAME span. A span past the break is never merged (its parallel BFS + // result is discarded), exactly as the sequential loop never processed it. + if (reachable.size > nodeBudget) { + truncatedByNodeCap = true; + break hopLoop; + } + const { exceeded, seeds: calleeSeeds } = spanSeeds[si]; + if (exceeded) truncatedByLimit = true; + if (calleeSeeds.length === 0) continue; + // A callee seed block IS reachable (the callee is invoked from the slice), + // unlike the top-level seed which is the target itself. + for (const id of calleeSeeds) { + if (!visited.has(id)) { + visited.add(id); + reachable.add(id); + hopReached.add(id); + } + } + const bfs = spanBfs[si]; + if (bfs === null) continue; + if (bfs.truncatedByLimit) truncatedByLimit = true; + // The per-callee BFS ran against a clone, so fold its discovered blocks + // into the shared `visited`/`reachable` here (the sequential path did this + // inside the BFS); Sets dedup, so order across siblings is irrelevant. + for (const id of bfs.reachable) { + visited.add(id); + reachable.add(id); + hopReached.add(id); + } + } + + if (reachable.size > nodeBudget) { + truncatedByNodeCap = true; + break; + } + sliceBlocks = [...hopReached]; + } + // Frontier of callees still expandable after the hop budget ⇒ depth truncation. + // (Conservative: if the last hop reached blocks AND we used the full budget, + // deeper callees may exist.) + if (hopsReached >= depthBudget && sliceBlocks.length > 0) truncatedByDepth = true; + + return { + reachable, + hopsReached, + truncatedByDepth, + truncatedByLimit, + truncatedByNodeCap, + ascentBlocks, + }; +} + +export interface RunPdgImpactDeps { + repo: { lbugPath: string }; + sym: { id: string; name: string; filePath: string; startLine?: number; endLine?: number }; + symType: string; + direction: 'upstream' | 'downstream'; + maxDepth: number; + limit: number; + /** Statement anchor (1-based source line). */ + line?: number; + executeParameterized: typeof executeParameterized; + /** + * Whether this index carries the FU-C `CALL_SUMMARY` return-value-ascent layer + * (from `meta.pdg.hasCallSummary`, read by the caller's `pdgLayerStatus`). + * `false`/`undefined` ⇒ a PRE-FU-C (v3) `--pdg` index: the intra slice is + * served unchanged, but the result `note` flags "no return-value ascent + * (re-index for CALL_SUMMARY)". Defaults to a sound `false` (no false ascent + * claim) when the caller omits it. + */ + callSummaryAvailable?: boolean; +} + +export async function runImpactPDG(deps: RunPdgImpactDeps): Promise { + const { repo, sym, direction, maxDepth, line, executeParameterized: exec } = deps; + const callSummaryAvailable = deps.callSummaryAvailable === true; + // `line` present ⇒ statement-anchored slice (the useful mode); absent ⇒ + // whole-symbol seed (intra-procedural reach collapses to empty for a + // function — kept for back-compat, with a note steering the caller to `line`). + const statementMode = typeof line === 'number' && Number.isInteger(line) && line >= 1; + // `target` carries the call-graph-compatible shape (id/name/type/filePath) so + // `collectImpactSymbolUids` keys on it identically to a callgraph result. + const target = { + id: sym.id, + name: sym.name, + type: deps.symType || 'Function', + filePath: sym.filePath, + }; + + // Validate the per-step LIMIT as a positive integer (KTD11 — interpolated, + // so it must be sanitised, never user-string-passed). A non-integer / out-of + // range value (NaN, 1.5, negative, huge) is CLAMPED to the bounded default + // rather than rejected: impact's `limit` is a soft page hint, and a clamp + // keeps the safety tool producing a (flagged-bounded) radius instead of a + // hard error. The clamp ceiling matches `pdg_query`'s validated max. + const rawLimit = deps.limit; + const stepLimit = + Number.isInteger(rawLimit) && rawLimit >= 1 && rawLimit <= PDG_QUERY_MAX_LIMIT + ? rawLimit + : PDG_QUERY_DEFAULT_LIMIT; + // Depth: clamp to the documented impact server max. The BFS issues one DB + // query per depth level, so direct callTool callers must not bypass the + // schema's maxDepth cap. + const depthBudget = + Number.isInteger(maxDepth) && maxDepth >= 1 ? Math.min(maxDepth, IMPACT_MAX_DEPTH) : 3; + + // ── Seed: anchor the target's BasicBlocks from the ALREADY-RESOLVED symbol ─ + // `_impactImpl` already resolved `sym` to a confident single match honoring + // the caller's target_uid/file_path/kind hints. Re-resolving by the bare + // `sym.name` here would RE-AMBIGUATE a disambiguated name (returning the + // "ambiguous" early payload instead of the PDG result) or anchor the seed on + // a DIFFERENT same-name symbol in another file (wrong-symbol blast radius). + // So build the seed anchor DIRECTLY from the resolved symbol's + // [startLine+1, endLine+1] window — the same window `resolveBlockAnchor`'s + // symbol branch produces, without re-running `resolveSymbolCandidates`. + const { anchorClause, queryParams } = statementMode + ? blockAnchorForStatement(sym, line as number) + : blockAnchorForResolvedSymbol(sym); + + const probeLimit = stepLimit + 1; + const rawSeedRows = await exec( + repo.lbugPath, + `MATCH (a:BasicBlock) WHERE ${anchorClause} RETURN a.id AS id LIMIT ${probeLimit}`, + queryParams, + ); + const seedRows = rawSeedRows.slice(0, stepLimit) as Array>; + let seedBlocks: string[] = seedRows + .map((r) => String(r['id'] ?? '')) + .filter((id: string) => id.length > 0); + // Pin the OWNING function for a statement seed: a closure body block that + // starts on the SAME source line as the seeded statement satisfies the + // (forgiving) symbol-span window but belongs to a different function, so its + // intra slice would leak in. The block id encodes the 1-based function start + // line, and a block of THIS symbol has fnLine === sym.startLine + 1 (block + // lines 1-based, symbol startLine 0-based). Drop foreign-fn seed blocks — + // defensively: only when it leaves ≥1 seed, so a kind whose fnLine convention + // differs never loses a real seed (it keeps the prior, slightly-loose set). + if (statementMode && typeof sym.startLine === 'number') { + const ownerFnLine = sym.startLine + 1; + const owned = seedBlocks.filter((id) => fnLineOf(id) === ownerFnLine); + if (owned.length > 0) seedBlocks = owned; + } + // FIX 7: the seed query probes one row past `stepLimit`, then processes at + // most `stepLimit` rows like every BFS step. A function with more seed blocks + // than `stepLimit` would silently under-seed (and thus under-report) — flag + // it so the result carries the same truncation + // signal the BFS steps do, never a silent partial seed. + const seedTruncated = rawSeedRows.length > stepLimit; + + // ── KTD6 no-body contract: distinguish "no PDG body" from "no dependence" ── + // A symbol that resolves but produces ZERO anchored blocks has no CFG body + // (interface / type alias / abstract / ambient / one-line const). A bare + // impactedCount:0 / risk:'LOW' would read as "safe to refactor" — the exact + // false-safe `impact` exists to prevent (#2129/#1858). Surface an explicit + // note + a non-LOW epistemic marker, never a silent confident zero. + if (seedBlocks.length === 0) { + return { + mode: 'pdg', + pdgResultVersion: PDG_RESULT_VERSION, + target, + direction, + ...(statementMode ? { criterionLine: line } : {}), + reachableBlocks: [], + intraReachableBlocks: [], + seedBlocks: [], + blockCount: 0, + affectedStatements: [], + affectedStatementCount: 0, + truncated: false, + depthReached: 0, + // statementMode: the requested line has no statement block inside the + // symbol (blank line, comment, outside the body, or a line the CFG did + // not materialise). Distinct from "no PDG body". + epistemic: statementMode ? 'pdg-no-block-at-line' : 'no-pdg-body', + note: statementMode + ? `No PDG statement block starts at line ${line} within '${sym.name}' ` + + `(${sym.filePath}). The line may be blank, a comment, a brace, or outside ` + + `the symbol's body. Pass a line that begins an executable statement.` + : `'${sym.name}' has no PDG body — no BasicBlocks / control- or data-dependence ` + + `edges exist for this symbol (e.g. an interface, type alias, abstract/ambient ` + + `member, or a one-line declaration with no CFG). This is NOT a confident ` + + `"no impact": the local PDG statement slice cannot model this symbol kind. ` + + `Inter-procedural symbol reach may still be attached by the unified impact dispatcher.`, + impactedCount: 0, + risk: 'UNKNOWN', + // KTD8 parity fields so a consumer iterating byDepth / reading the + // depth counts on a no-body result still finds a well-formed (empty) + // shape rather than `undefined` (which would render as "isolated"). + ...emptyPdgParityFields(), + unresolvedBlockCount: 0, + ambiguousProjectionCount: 0, + }; + } + + // ── Bounded direction-aware BFS over CDG + REACHING_DEF (KTD4, KTD11) ────── + // Seed blocks are NOT counted as reachable (they ARE the target); the + // reachable set is everything the BFS discovers from them. Visited tracks + // BOTH seeds and discovered blocks so a cycle never re-expands. The traversal + // is the shared `bfsReachableBlocks` — the SAME edge query / step-limit / probe + // semantics the inter-procedural descent (U1) reuses, so the two cannot diverge. + const visited = new Set(seedBlocks); + const intra = await bfsReachableBlocks({ + lbugPath: repo.lbugPath, + exec, + seedBlocks, + visited, + direction, + depthBudget, + stepLimit, + probeLimit, + }); + const reachable = intra.reachable; + // FIX 6: snapshot the INTRA-procedural reachable set BEFORE the U1 descent + // expands `reachable` with inter-procedurally-reached callee blocks. The + // callgraph bridge keys its "first-hop proven" set on seed ∪ intra-reachable + // (the original statement slice) — feeding it the interproc-expanded superset + // would mark transitively-reached (2+ hop) callgraph targets as first-hop + // proven, silently shifting the established statementPrecision semantics. + const intraReachableBlocks = [...reachable].sort(); + const depthReached = intra.depthReached; + // `truncatedByDepth`: the BFS still had a non-empty frontier when the depth + // budget ran out (more reachable blocks exist past `maxDepth`). + // `truncatedByLimit`: a step's neighbour query hit the one-past LIMIT probe. + // The SEED query is one-past-probed too, so `seedTruncated` seeds the flag — a + // partial seed is a lower-bound expansion just like a partial step. + let truncatedByDepth = intra.truncatedByDepth; + let truncatedByLimit = seedTruncated || intra.truncatedByLimit; + + // ── U1: bounded inter-procedural forward closure (DOWNSTREAM only) ────────── + // After the intra slice completes and BEFORE block→symbol projection, descend + // through resolved call sites so the slice crosses function boundaries (HRB + // context-insensitive forward closure — Joern's shipped approach). Gated to + // downstream (the dispatcher's direction gate mirrors this): forward closure is + // only meaningful when following calls forward. A pre-namespace-v4 index (no + // `calleeIds`) yields no callee ids → the descent is a no-op → byte-identical + // intra-only behavior. The inter-procedural reach DEEPENS `reachableBlocks` + // (and thus `affectedStatements`); the owning-symbol `byDepth` stays a single + // collapsed bucket (block-hops are not call-hops — the standing KTD8 contract; + // every byDepth consumer iterates generically, so deepening the statement slice + // — not the bucket count — is the correctness-preserving surface). + let interproceduralHops = 0; + // U-C4: CALL blocks whose callee's CALL_SUMMARY licensed a return-value ascent + // (empty for an upstream slice, a pre-FU-C v3 index, or no return-flowing + // callee). The statement projection expands these blocks to their interior + // call lines (a coalesced call block spans several statements whose results + // chain through it — the statement-granularity realisation of the ascent). + let ascentBlocks = new Set(); + if (direction === 'downstream') { + const interproc = await interproceduralDescent({ + lbugPath: repo.lbugPath, + exec, + // The slice the first hop gathers callees from = seeds ∪ intra reach. The + // seed block(s) are included because a callee invoked directly on the + // changed line is the most-directly-impacted one (seed-minus-reachable + // excludes seeds from `reachable`, but they still call out to callees). + initialSliceBlocks: [...seedBlocks, ...reachable], + visited, + depthBudget: INTERPROC_DEPTH_BUDGET, + // FIX 1: the per-callee intra BFS depth is the SAME depth clamp the + // top-level intra BFS uses — NOT `stepLimit` (a row/probe budget). A callee + // must not be traversable up to PDG_QUERY_MAX_LIMIT dependence-levels deep. + intraDepthBudget: depthBudget, + stepLimit, + probeLimit, + nodeBudget: INTERPROC_NODE_BUDGET, + // FIX 3: pre-seed the descent with the statement seed's OWNER FUNCTION so + // direct/mutual recursion back to it never re-seeds the whole-function span + // the statement slice deliberately excluded. The resolved target symbol id + // (`sym.id`) is exactly that owner — `_impactImpl` already resolved it. + preSeededCalleeIds: statementMode ? [sym.id] : [], + }); + interproceduralHops = interproc.hopsReached; + ascentBlocks = interproc.ascentBlocks; + for (const id of interproc.reachable) reachable.add(id); + if (interproc.truncatedByDepth) truncatedByDepth = true; + if (interproc.truncatedByLimit) truncatedByLimit = true; + // FIX 7: the node cap is a SIZE/budget limit, semantically 'limit' — NOT + // 'depth' (which means dependence-level exhaustion). Map it to truncatedByLimit. + if (interproc.truncatedByNodeCap) truncatedByLimit = true; + // FIX 5: do NOT fold the inter-procedural FUNCTION-hop count into + // `depthReached` (which means intra BLOCK-hop depth — a different unit). The + // function-hop count is plumbed separately via `interproceduralHops`. + } + + const reachableBlocks = [...reachable].sort(); + const truncated = truncatedByDepth || truncatedByLimit; + // truncatedBy PRECEDENCE (U3): when BOTH depth and limit truncation fire, the + // scalar `truncatedBy` reports 'depth' (the depth ternary is tested first), while + // `truncatedByReasons` lists BOTH ['depth','limit']. Depth wins the scalar slot + // because dependence-level exhaustion is the stronger "the slice is incomplete" + // signal; the reasons array preserves that a size/limit cap also fired. This is a + // standing precedence contract — keep the depth-first ternary and the both-fire + // reasons array consistent. + const truncatedBy: 'depth' | 'limit' | undefined = truncatedByDepth + ? 'depth' + : truncatedByLimit + ? 'limit' + : undefined; + const truncatedByReasons: readonly ('depth' | 'limit')[] | undefined = + truncatedByDepth && truncatedByLimit ? (['depth', 'limit'] as const) : undefined; + + // ── Resolve the reachable blocks to source statements (line + text) ──────── + // This is the useful output of statement mode: the dependent statements the + // change at `line` reaches. Fetched once for the whole reachable set; sorted + // by line. Failure surfaces (no `.catch` swallow) rather than masquerading + // as "no affected statements". + // Scope tag (FU-A): the criterion's OWN function is (sym.filePath, fnLine), + // where fnLine follows the BasicBlock 1-based convention sym.startLine + 1 + // (the same window used to anchor the seed above). A symbol without a numeric + // startLine has no owning-fn line to match, so `ownerFnLine` is NaN and every + // statement tags as 'inter' (no false intra claim). + const criterionFile = sym.filePath; + const ownerFnLine = typeof sym.startLine === 'number' ? sym.startLine + 1 : Number.NaN; + // ── FU-B-2 chain-walk blocks: statement-granular interior recovery ───────── + // The SINGLE principled mechanism that expands a coalesced straight-line + // BasicBlock to its interior dependent statements via the self REACHING_DEF + // def→use LINE chain. Two contributors: + // - the SEED block(s), seeded at the criterion `line` (statement mode only): + // recovers the interior statements of the criterion's own coalesced block + // that the block-granular slice lost (the U2 intra-dataflow-chain gap — + // line 7's `7-9` block yields {8,9}). The seed block is NOT in + // `reachableBlocks` (seeds are excluded), so the walk surfaces lines no + // other projection reaches. + // - each U-C4 ascent CALL block, seeded from its own start line (empty entry + // set ⇒ the projection default-seeds at the block start): the principled + // replacement for FU-C's blind all-interior-lines expansion. + const chainWalkBlocks = new Map>(); + if (statementMode) { + for (const id of seedBlocks) chainWalkBlocks.set(id, new Set([line as number])); + } + // Every REACHABLE coalesced block is also walked, seeded from its own start + // line (empty entry set ⇒ the projection default-seeds at the block start): a + // block reached via CDG (a control-dependent body block) or REACHING_DEF can + // itself coalesce a straight-line data-dep chain whose interior statements the + // block-start projection would lose (e.g. the guard fixture's `const y; const + // z = y + 1;` post-guard body — line 12 chains from line 11 via `y`). This is + // the SAME principled self-edge walk, applied to the whole slice, not just the + // seed. + for (const id of reachableBlocks) { + if (!chainWalkBlocks.has(id)) chainWalkBlocks.set(id, new Set()); + } + for (const id of ascentBlocks) { + if (!chainWalkBlocks.has(id)) chainWalkBlocks.set(id, new Set()); + } + const affectedStatements = await pdgStatementsForBlocks( + repo.lbugPath, + reachableBlocks, + exec, + criterionFile, + ownerFnLine, + chainWalkBlocks, + ); + + // ── Has a PDG body but no inter-block dependence reachability ────────────── + // Distinct from "no PDG body": the function exists and has blocks, but no + // CDG/REACHING_DEF edge leaves the target's blocks in this direction (no + // DISTINCT downstream BasicBlock is reached). For a WHOLE-SYMBOL seed this is + // the expected (and uninformative) result — every intra-procedural block is + // already a seed — so the note steers to `line`. Still not a confident zero — + // explicit note + UNKNOWN (KTD6/KTD8). + // + // Item 5 robustness: `affectedStatements` was ALREADY computed above with the + // seed block(s) in `chainWalkBlocks`, so a criterion whose only dependents are + // INTERIOR to its own coalesced seed block (a straight-line chain with no + // distinct downstream block — e.g. `const a; const b = a*2; const c = b-3` + // with no `return`) still has those recovered interior statements here. Surface + // them rather than hardcoding `[]` (which would silently drop the seed-block + // chain-walk on this path). `reachableBlocks` stays `[]` — those lines belong to + // the seed block, correctly not a reachable BLOCK — but the statement slice is + // not lost. + if (reachableBlocks.length === 0) { + return { + mode: 'pdg', + pdgResultVersion: PDG_RESULT_VERSION, + target, + direction, + ...(statementMode ? { criterionLine: line } : {}), + impactedCount: 0, + risk: 'UNKNOWN', + epistemic: 'pdg-intra-procedural', + note: statementMode + ? `No statement in '${sym.name}' is in a DISTINCT ${direction}-dependent block of ` + + `line ${line} (no CDG/REACHING_DEF edge leaves the seed block in this direction). ` + + `Any interior statements of the seed's own coalesced block that depend on line ${line} ` + + `are surfaced in affectedStatements.` + : `'${sym.name}' has a PDG body but a WHOLE-SYMBOL ${direction} slice is empty: ` + + `intra-procedural dependence stays inside the function, so every reachable block ` + + `is already part of the seed. Pass line: to slice from a specific statement ` + + `(what depends on the code at that line). Inter-procedural symbol reach is attached ` + + `separately by the unified impact dispatcher.`, + reachableBlocks: [] as string[], + // Empty intra reach (reachableBlocks is empty here) — the bridge keys on + // seed ∪ intra-reachable, so an empty intra slice leaves the bridge to seed + // from the seed blocks alone (FIX 6 parity field). + intraReachableBlocks: [] as string[], + // Carry the real seed blocks (non-empty here — the function HAS blocks, they + // are all seeds): a callee invoked directly on the seeded line must still be + // provable even when the line has no downstream dependents (the seed-line FN + // the tri-review found). Empty reachableBlocks must NOT zero the seed callees. + seedBlocks, + blockCount: 0, + affectedStatements, + affectedStatementCount: affectedStatements.length, + depthReached, + unresolvedBlockCount: 0, + ambiguousProjectionCount: 0, + ...(truncated ? { truncated: true } : {}), + ...(truncatedBy ? { truncatedBy } : {}), + ...(truncatedByReasons ? { truncatedByReasons } : {}), + ...emptyPdgParityFields(), + }; + } + + // ── U4: project reachable blocks → owning symbols, assemble parity result ── + const projection = await projectBlocksToSymbols({ + lbugPath: repo.lbugPath, + blockIds: reachableBlocks, + executeParameterized: exec, + }); + + return assemblePdgImpactResult({ + target: { + id: sym.id, + name: sym.name, + type: deps.symType || 'Function', + filePath: sym.filePath, + }, + direction, + reachableBlocks, + intraReachableBlocks, + seedBlocks, + affectedStatements, + criterionLine: statementMode ? (line as number) : undefined, + projection, + depthReached, + truncated, + truncatedBy, + truncatedByReasons, + interproceduralHops, + callSummaryAvailable, + }); +} + +// ── PDG inter-procedural bridge evidence + unified result composition ───────── +// Moved from local-backend.ts (#2227 U13): these are pure PDG-result-shaping +// helpers, cohesive with this module's projection/assembly role. LocalBackend +// keeps the DB-access seams (calleesOfBlocks, _runImpactBFS) and imports these. + +type PdgBridgeEvidence = Extract; + +export interface PdgBridgeEvidenceInfo { + evidence: PdgBridgeEvidence; + basis: string; +} + +export interface PdgBridgeOptions { + /** + * Leaf callee names invoked in the criterion's dependence-slice blocks + * (`BasicBlock.callees`). A first-hop callee is "proven" statement-precise iff + * its name is in this set — i.e. it is actually called from a statement the + * changed line reaches, not merely somewhere in the whole function. Empty/absent + * ⇒ no statement slice to discriminate (upstream or whole-symbol) ⇒ the symbol + * graph is used as a compatibility bridge (all callgraph-bridge), preserving + * callgraph reach. + * + * REQUIRED (co-populated with {@link sliceCalleeIds}): `local-backend` builds + * the bridge by reading BOTH cells from the same slice blocks, so a missing + * key is an EMPTY set, never `undefined`. Keeping both non-optional is what + * lets the capped-block sentinel guard below read `sliceCalleeNames` directly + * (the cap is name-agnostic — a capped block always carries the sentinel here). + */ + sliceCalleeNames: ReadonlySet; + /** + * Resolved callee symbol ids invoked in the criterion's dependence-slice blocks + * (`BasicBlock.calleeIds`). This is the SOUND primary key: a first-hop callee is + * "proven" statement-precise iff its resolved symbol id is in this set, which + * eliminates same-leaf-name collision (false-positive) and import-alias/rename + * (false-negative) — failure modes the name set cannot distinguish. Empty/absent + * ⇒ no captured ids (pre-v3 index / upstream / whole-symbol) ⇒ fall back to the + * leaf-name match (`sliceCalleeNames`). REQUIRED (co-populated with + * {@link sliceCalleeNames}) — an empty set means "no ids", never `undefined`. + */ + sliceCalleeIds: ReadonlySet; +} + +export function pdgBridgeEvidenceForImpact(input: { + bridge: PdgBridgeOptions; + depth: number; + calleeName: unknown; + calleeId?: unknown; + inherited?: PdgBridgeEvidenceInfo; +}): PdgBridgeEvidenceInfo { + const { bridge, depth, calleeName, calleeId, inherited } = input; + if (depth > 1) { + return ( + inherited ?? { + evidence: 'unproven-bridge', + basis: 'first-hop evidence unavailable for inherited symbol-graph reach', + } + ); + } + + const sliceCalleeNames = bridge.sliceCalleeNames; + // Whole-symbol compatibility bridge ONLY when NEITHER key discriminates. The + // empty-names guard must also require empty ids: `local-backend` builds the + // bridge when names OR ids have signal, so an id-only slice (names empty/absent, + // ids present — e.g. a block whose calls resolve to ids but carry no static leaf + // name) must fall through to the resolved-id branch, not short-circuit to + // "prove everything". (PR #2227 tri-review-2 headline.) + if (sliceCalleeNames.size === 0 && bridge.sliceCalleeIds.size === 0) { + return { + evidence: 'callgraph-bridge', + basis: 'whole-symbol PDG result uses symbol graph as compatibility bridge', + }; + } + + // A slice block whose call sites were truncated at the per-statement cap has an + // INCOMPLETE callee list, so absence from the set does not prove absence from + // the slice. Keep such reach callgraph-equal rather than under-proving. (A capped + // block always carries the sentinel in `callees`/names — the per-statement cap is + // name-agnostic — so checking names suffices. `sliceCalleeNames` is always + // present now (an id-only slice has it as an empty set, never capped), so the + // direct `.has` is sound.) + if (sliceCalleeNames.has(CALLEES_TRUNCATED_SENTINEL)) { + return { + evidence: 'callgraph-bridge', + basis: 'a slice block truncated its call sites — callee set is incomplete (callee-unknown)', + }; + } + + // Sound primary key (KTD3): when the slice carries resolved callee ids and the + // block is not capped (sentinel handled above), prove by exact symbol-id match. + // An id that is NOT in the present set is a real proof failure (`unproven-bridge`), + // NOT a fall-through to the name predicate — the name path would re-leak the + // same-name collision this key exists to eliminate. + const sliceCalleeIds = bridge.sliceCalleeIds; + if (sliceCalleeIds.size > 0) { + const id = typeof calleeId === 'string' ? calleeId : ''; + if (id && sliceCalleeIds.has(id)) { + return { + evidence: 'callgraph-bridge', + basis: + 'callee id is invoked in a block of the local PDG dependence slice (resolved-symbol match)', + }; + } + return { + evidence: 'unproven-bridge', + basis: 'callee id is not invoked in any block of the local PDG dependence slice', + }; + } + + // R3 graceful fallback: no captured ids (pre-v3 index / upstream / whole-symbol) + // ⇒ use the leaf-name match. + const name = typeof calleeName === 'string' ? calleeName : ''; + if (name && sliceCalleeNames.has(name)) { + return { + evidence: 'callgraph-bridge', + basis: 'callee is invoked in a block of the local PDG dependence slice', + }; + } + + return { + evidence: 'unproven-bridge', + basis: 'callee is not invoked in any block of the local PDG dependence slice', + }; +} + +/** + * Pick the stronger of two bridge-evidence verdicts for the same reached symbol. + * `callgraph-bridge` (proven) beats `unproven-bridge`, so a node reachable from + * multiple parents is proven if ANY parent proves it. This makes the + * proven/unproven label order-independent of DB row iteration — a diamond-reached + * symbol gets the same label regardless of which parent the BFS visits first + * (PR #2227 tri-review, P3). + */ +export function betterBridgeEvidence( + existing: PdgBridgeEvidenceInfo | undefined, + candidate: PdgBridgeEvidenceInfo, +): PdgBridgeEvidenceInfo { + if (!existing) return candidate; + if (existing.evidence === 'callgraph-bridge') return existing; + if (candidate.evidence === 'callgraph-bridge') return candidate; + return existing; +} + +function normalizePdgBridgeByDepth(byDepth: Record): Record { + const normalized: Record = {}; + for (const [depthKey, items] of Object.entries(byDepth ?? {})) { + const depth = Number(depthKey); + if (!Number.isFinite(depth) || !Array.isArray(items)) continue; + normalized[depth] = items.map((item) => { + if (!item || typeof item !== 'object') return item; + const record = item as Record; + const evidence = + typeof record.pdgEvidence === 'string' + ? (record.pdgEvidence as PdgImpactEvidence) + : 'callgraph-bridge'; + return { + ...record, + pdgEvidence: evidence, + ...(record.pdgEvidenceReason + ? {} + : { + pdgEvidenceReason: + evidence === 'unproven-bridge' + ? 'symbol reached through the resolved symbol graph, but the existing graph did not prove the first-hop call site is in the local PDG slice' + : 'symbol reached through the resolved symbol graph compatibility bridge', + }), + }; + }); + } + return normalized; +} + +function countPdgEvidence( + byDepth: Record, +): Partial> { + const counts: Partial> = {}; + for (const items of Object.values(byDepth ?? {})) { + if (!Array.isArray(items)) continue; + for (const item of items) { + if (!item || typeof item !== 'object') continue; + const evidence = (item as { pdgEvidence?: unknown }).pdgEvidence; + if (typeof evidence !== 'string') continue; + counts[evidence as PdgImpactEvidence] = (counts[evidence as PdgImpactEvidence] ?? 0) + 1; + } + } + return counts; +} + +function dominantInterproceduralEvidence( + counts: Partial>, +): PdgBridgeEvidence | undefined { + if ((counts['unproven-bridge'] ?? 0) > 0) return 'unproven-bridge'; + if ((counts['callgraph-bridge'] ?? 0) > 0) return 'callgraph-bridge'; + return undefined; +} + +/** + * Statement-precise projection of the inter-procedural bridge: the subset PROVEN + * to be invoked from the criterion's dependence slice (`callgraph-bridge`), + * dropping `unproven-bridge` symbols — reachable in the call graph but only from + * statements the changed line does not reach. Additive: the full + * `interproceduralByDepth` is unchanged and still preserves callgraph reach; this + * answers the tighter "which other functions does changing THIS line affect?". + * For an upstream / whole-symbol seed there is no discriminating slice, so every + * symbol is `callgraph-bridge` and the projection equals the full reach + * (`statementPrecision` = 1). When the slice carries resolved callee ids the + * projection is SOUND — a reached symbol is proven iff its resolved id matches a + * slice-block id (resolved-symbol match), so neither a same-named out-of-slice + * callee nor an aliased import perturbs the set. The leaf-name match is the + * documented fallback (pre-v3 index / no captured ids), and only in that + * name-fallback path is the projection a conservative SUPERSET (a callee invoked + * from both an in-slice and out-of-slice site resolves to proven). + */ +function projectStatementPreciseByDepth( + byDepth: Record, +): Record { + const out: Record = {}; + for (const [depthKey, items] of Object.entries(byDepth ?? {})) { + const depth = Number(depthKey); + if (!Number.isFinite(depth) || !Array.isArray(items)) continue; + const proven = items.filter( + (item) => + item && + typeof item === 'object' && + (item as { pdgEvidence?: unknown }).pdgEvidence !== 'unproven-bridge', + ); + if (proven.length > 0) out[depth] = proven; + } + return out; +} + +/** + * Compose `mode:'pdg'` into one user-facing impact result: + * + * - `affectedStatements` / `reachableBlocks` stay owned by the persisted PDG + * layer (CDG + REACHING_DEF), preserving the statement-level intra result. + * - `interproceduralByDepth` / `pdgInterprocedural` expose the symbol reach; + * `byDepth` stays as the compatibility symbol bucket so existing consumers + * still see one PDG result shape. + * + * The callgraph option remains available as the comparator/default path; this + * helper only changes the `pdg` result contract from intra-only to unified. + */ +export function composeUnifiedPdgImpactResult( + pdgResult: PdgImpactResult, + interproceduralResult: any | null, + interproceduralError?: unknown, +): PdgImpactResult { + if ('error' in pdgResult || 'pdgLayer' in pdgResult) return pdgResult; + + const localByDepth = pdgResult.byDepth ?? {}; + const localByDepthCounts = pdgResult.byDepthCounts ?? {}; + const interproceduralByDepth = normalizePdgBridgeByDepth(interproceduralResult?.byDepth ?? {}); + const interproceduralByDepthCounts = interproceduralResult?.byDepthCounts ?? {}; + const interproceduralEvidenceCounts = countPdgEvidence(interproceduralByDepth); + const interproceduralEvidence = dominantInterproceduralEvidence(interproceduralEvidenceCounts); + // Additive statement-precise projection (see projectStatementPreciseByDepth): + // the proven subset of the inter-procedural reach. `interproceduralByDepth` + // above is unchanged and still preserves full callgraph reach. + const statementPreciseByDepth = projectStatementPreciseByDepth(interproceduralByDepth); + const statementPreciseByDepthCounts: Record = {}; + for (const [depthKey, items] of Object.entries(statementPreciseByDepth)) { + statementPreciseByDepthCounts[Number(depthKey)] = (items as unknown[]).length; + } + const provenBridgeCount = interproceduralEvidenceCounts['callgraph-bridge'] ?? 0; + const unprovenBridgeCount = interproceduralEvidenceCounts['unproven-bridge'] ?? 0; + const statementPrecision = + provenBridgeCount + unprovenBridgeCount > 0 + ? provenBridgeCount / (provenBridgeCount + unprovenBridgeCount) + : null; + const byDepth: Record = {}; + const byDepthCounts: Record = {}; + const depthKeys = Array.from( + new Set([ + ...Object.keys(localByDepth), + ...Object.keys(interproceduralByDepth), + ...Object.keys(localByDepthCounts), + ...Object.keys(interproceduralByDepthCounts), + ]), + ) + .map((d) => Number(d)) + .filter((d) => Number.isFinite(d)) + .sort((a, b) => a - b); + // Resolved symbol id of a byDepth item, or null for an unresolved/shadow item. + const itemId = (item: unknown): string | null => { + if (item && typeof item === 'object' && 'id' in item) { + const id = (item as { id?: unknown }).id; + return typeof id === 'string' && id.length > 0 ? id : null; + } + return null; + }; + // Cross-bucket dedup: a callee can surface in BOTH the local PDG block-expansion + // AND the inter-procedural callgraph reach, so the raw `local + interproc` sums + // double-count it. Track each layer's resolved ids to subtract the overlap from + // the headline (cross-depth) and the per-depth counts. + const localAllIds = new Set(); + const interAllIds = new Set(); + for (const depth of depthKeys) { + const localItems = localByDepth[depth] ?? localByDepth[String(depth)] ?? []; + const interItems = interproceduralByDepth[depth] ?? interproceduralByDepth[String(depth)] ?? []; + const localDepthIds = new Set(); + for (const it of localItems) { + const id = itemId(it); + if (id !== null) { + localDepthIds.add(id); + localAllIds.add(id); + } + } + // Drop an interproc item whose resolved id is already present locally AT THIS + // DEPTH — keep the local item (richer projection). The SAME id reached at a + // DIFFERENT depth is a legitimate distinct per-depth bucket and is retained + // (so sum(byDepthCounts) can exceed the cross-depth `impactedCount`). + let perDepthOverlap = 0; + const dedupedInter: unknown[] = []; + for (const it of interItems) { + const id = itemId(it); + if (id !== null) interAllIds.add(id); + if (id !== null && localDepthIds.has(id)) { + perDepthOverlap += 1; + continue; + } + dedupedInter.push(it); + } + const items = [...localItems, ...dedupedInter]; + if (items.length > 0) byDepth[depth] = items; + const localCount = + localByDepthCounts[depth] ?? localByDepthCounts[String(depth)] ?? localItems.length; + const interCount = + interproceduralByDepthCounts[depth] ?? + interproceduralByDepthCounts[String(depth)] ?? + interItems.length; + const totalCount = Math.max(0, localCount + interCount - perDepthOverlap); + if (totalCount > 0) byDepthCounts[depth] = totalCount; + } + // Cross-depth overlap of resolved ids reached by BOTH layers. Computed from the + // visible byDepth ids; if a layer's byDepth was display-truncated this is a + // lower bound, so the resulting count is a safe over-estimate (never the old + // double-count, never below the larger single layer). + let crossOverlap = 0; + for (const id of interAllIds) if (localAllIds.has(id)) crossOverlap += 1; + + if (Object.keys(byDepthCounts).length === 0) { + const localZero = localByDepthCounts[1] ?? localByDepthCounts['1']; + const interZero = interproceduralByDepthCounts[1] ?? interproceduralByDepthCounts['1']; + if (typeof localZero === 'number' || typeof interZero === 'number') { + byDepthCounts[1] = + (typeof localZero === 'number' ? localZero : 0) + + (typeof interZero === 'number' ? interZero : 0); + } + } + + const localImpactedCount = + typeof pdgResult.impactedCount === 'number' ? pdgResult.impactedCount : 0; + const interproceduralImpactedCount = + typeof interproceduralResult?.impactedCount === 'number' + ? interproceduralResult.impactedCount + : 0; + // DISTINCT owning symbols across both layers (was localImpactedCount + + // interproceduralImpactedCount, which double-counted a symbol reached by both). + const impactedCount = Math.max( + 0, + localImpactedCount + interproceduralImpactedCount - crossOverlap, + ); + // `direct` = distinct depth-1 reach. The deduped byDepthCounts[1] is exactly + // that union, so it stays consistent with impactedCount (no separate layer sum). + const directCount = byDepthCounts[1] ?? pdgResult.summary?.direct ?? localImpactedCount; + const summary = interproceduralResult?.summary + ? { + ...interproceduralResult.summary, + direct: directCount, + } + : { + direct: directCount, + processes_affected: 0, + modules_affected: 0, + }; + const affectedProcesses = interproceduralResult?.affected_processes ?? []; + const affectedModules = interproceduralResult?.affected_modules ?? []; + const partial = Boolean(interproceduralResult?.partial || interproceduralError); + const errorMessage = + interproceduralError instanceof Error + ? interproceduralError.message + : interproceduralError + ? String(interproceduralError) + : undefined; + + const noteParts = [ + pdgResult.note, + `Inter-procedural symbol reach is included using the resolved symbol graph; ` + + `statement-level PDG reach remains in affectedStatements. The symbol reach is ` + + `labeled as a PDG evidence bridge, not as pure statement-level dependence.`, + ]; + if (errorMessage) { + noteParts.push( + `Inter-procedural symbol reach failed (${errorMessage}); byDepth is therefore a lower bound.`, + ); + } else if (interproceduralResult?.epistemic === 'lower-bound') { + noteParts.push( + `The inter-procedural symbol reach is a lower bound because unresolved indirection was detected.`, + ); + } + if ((interproceduralEvidenceCounts['unproven-bridge'] ?? 0) > 0) { + noteParts.push( + `${interproceduralEvidenceCounts['unproven-bridge']} inter-procedural ` + + `symbol(s) are labeled unproven-bridge: the resolved symbol graph reaches them, ` + + `but the current graph did not prove their first-hop call site is in the local PDG slice.`, + ); + } + + return { + ...pdgResult, + // Explicit (also carried by the `...pdgResult` spread) so the unified + // mode:'pdg' exit always advertises the contract version. + pdgResultVersion: PDG_RESULT_VERSION, + impactedCount, + note: noteParts.filter(Boolean).join(' '), + summary, + byDepthCounts, + interproceduralByDepth, + interproceduralByDepthCounts, + // Statement-precise (proven) inter-procedural reach is emitted ONLY under + // `pdgInterprocedural` below — a single source, no top-level duplicate. + affected_processes: affectedProcesses, + affected_modules: affectedModules, + byDepth, + ...(partial ? { partial: true } : {}), + ...(interproceduralResult?.epistemic + ? { interproceduralEpistemic: interproceduralResult.epistemic } + : {}), + ...(interproceduralResult?.boundaries + ? { interproceduralBoundaries: interproceduralResult.boundaries } + : {}), + ...(errorMessage ? { interproceduralError: errorMessage } : {}), + pdgEvidence: { + // pdgResult is narrowed to the success/empty slice result by the + // `'error' in / 'pdgLayer' in` guard at the top of this function, so + // `pdgEvidence` is typed (optional) — no `as any`. + ...(pdgResult.pdgEvidence ?? {}), + ...(interproceduralEvidence ? { interprocedural: interproceduralEvidence } : {}), + interproceduralEvidenceCounts, + }, + pdgInterprocedural: { + engine: 'symbol-graph', + evidence: interproceduralEvidence ?? 'callgraph-bridge', + impactedCount: interproceduralImpactedCount, + byDepthCounts: interproceduralByDepthCounts, + byDepth: interproceduralByDepth, + evidenceCounts: interproceduralEvidenceCounts, + statementPreciseByDepth, + statementPreciseByDepthCounts, + statementPreciseImpactedCount: provenBridgeCount, + statementPrecision, + partial, + }, + }; +} diff --git a/gitnexus/src/mcp/tools.ts b/gitnexus/src/mcp/tools.ts index 959f6a565..a585c867f 100644 --- a/gitnexus/src/mcp/tools.ts +++ b/gitnexus/src/mcp/tools.ts @@ -77,6 +77,10 @@ export const EXPLAIN_MAX_LIMIT = 200; export const PDG_QUERY_DEFAULT_LIMIT = 50; export const PDG_QUERY_MAX_LIMIT = 200; +// Shared impact traversal depth cap. The MCP schema advertises this bound; +// PDG direct backend callers also enforce it before running traversal. +export const IMPACT_MAX_DEPTH = 32; + export const GITNEXUS_TOOLS: ToolDefinition[] = [ { name: 'list_repos', @@ -409,11 +413,17 @@ Each edit is tagged with confidence: description: `Analyze the blast radius of changing a code symbol. Returns affected symbols grouped by depth, plus risk assessment, affected execution flows, and affected modules. +MODE (opt-in): "callgraph" (default) walks symbol→symbol edges (CALLS/IMPORTS/EXTENDS/IMPLEMENTS) — inter-procedural, the established comparator/default behavior. "pdg" requires an index built with \`gitnexus analyze --pdg\` and returns one unified PDG-facing result: statement-level control/data dependence from the persisted PDG plus inter-procedural symbol reach. The explicit interprocedural surface is interproceduralByDepth/pdgInterprocedural; byDepth remains the compatibility symbol bucket. pdg remains incompatible with crossDepth and @group targets; relationTypes/minConfidence filter the inter-symbol reach. + +STATEMENT-ANCHORED PDG SLICE: with mode:'pdg', pass "line" (1-based source line within the target symbol) to seed the dependence slice on the statement at that line and return what depends on it in affectedStatements (line + text). Inter-procedural symbols are still reported through interproceduralByDepth/pdgInterprocedural and the compatibility byDepth bucket. Without "line", pdg returns whole-symbol inter-procedural reach plus local whole-symbol PDG diagnostics. + +PDG OUTPUT CONTRACT: every mode:'pdg' result (success, empty, degraded, or error) carries pdgResultVersion:1 — a stable discriminator for external consumers that bumps on any breaking change to the PDG result shape (distinct from the DB schema version). Successful PDG results include mode:'pdg', a full target envelope (id/name/type/filePath), affectedStatements, affectedStatementCount, interproceduralByDepth/pdgInterprocedural for cross-function reach, compatibility byDepth/byDepthCounts, risk:'UNKNOWN', and a note describing the unified contract. Degraded PDG results (no-layer, sub-layer-missing, unknown) keep mode:'pdg', pdgResultVersion:1, target metadata when the target resolves, risk:'UNKNOWN', note/remediation, and empty byDepth parity fields — never a false-safe zero. If depth and limit both bound the slice, truncatedByReasons reports both causes while truncatedBy remains scalar. + WHEN TO USE: Before making code changes — especially refactoring, renaming, or modifying shared code. Shows what would break. AFTER THIS: Review d=1 items (WILL BREAK). Use context() on high-risk symbols. Output includes: -- risk: LOW / MEDIUM / HIGH / CRITICAL +- risk: LOW / MEDIUM / HIGH / CRITICAL / UNKNOWN - summary: direct callers, processes affected, modules affected - affected_processes: which execution flows break and at which step - affected_modules: which functional areas are hit (direct vs indirect) @@ -450,6 +460,19 @@ SERVICE: optional monorepo path prefix (case-sensitive path segments). When "rep type: 'string', description: 'upstream (what depends on this) or downstream (what this depends on)', }, + mode: { + type: 'string', + enum: ['callgraph', 'pdg'], + default: 'callgraph', + description: + "Blast-radius engine. 'callgraph' (default) = inter-procedural symbol→symbol traversal (established comparator). 'pdg' = unified PDG-facing impact: intra-procedural statement-level affectedStatements from the persisted control/data dependence layer plus inter-procedural symbols in interproceduralByDepth/pdgInterprocedural and the compatibility byDepth bucket; requires `gitnexus analyze --pdg`. PDG symbol reach is labeled as a PDG evidence bridge, not pure statement-level dependence, and successful PDG results are UNKNOWN-risk. PDG is incompatible with crossDepth and @group targets; relationTypes/minConfidence filter the inter-symbol reach.", + }, + line: { + type: 'integer', + minimum: 1, + description: + "1-based source line — PDG statement anchor (mode:'pdg'). Seeds affectedStatements on the statement at this line; inter-procedural symbols are still returned in interproceduralByDepth/pdgInterprocedural and the compatibility byDepth bucket.", + }, file_path: { type: 'string', description: 'File path hint to disambiguate common names', @@ -464,7 +487,7 @@ SERVICE: optional monorepo path prefix (case-sensitive path segments). When "rep description: 'Max relationship depth (default: 3, server clamps to 1–32)', default: 3, minimum: 1, - maximum: 32, + maximum: IMPACT_MAX_DEPTH, }, crossDepth: { type: 'number', diff --git a/gitnexus/src/storage/parse-cache.ts b/gitnexus/src/storage/parse-cache.ts index 5061e635a..c460a480d 100644 --- a/gitnexus/src/storage/parse-cache.ts +++ b/gitnexus/src/storage/parse-cache.ts @@ -185,15 +185,27 @@ export const computeChunkHash = ( // default-cap runs share a key. The emit-time edge cap is deliberately // absent — see the PdgCacheKey doc comment. // - // NAMESPACE VERSION (`pdg:2`): bumped when the worker-emitted + // NAMESPACE VERSION (`pdg:5`): bumped when the worker-emitted // `cfgSideChannel` SHAPE changes for pdg-mode runs only — pdg:1→2 in #2083 - // M3 U1 (TsHarvester emits taint `sites` on StatementFacts). Invalidates - // pdg-mode chunks and their durable parsedfile-cache entries; flag-off - // chunk keys never reach this line and stay byte-identical, so non-pdg - // users pay nothing. Deliberately NOT a SCHEMA_BUMP — that gates the whole - // cache version and would force a full cold re-parse on EVERY user (the M1 - // bump comment above records that cost). - const ns = `pdg:2;maxFn=${opts.maxFunctionLines ?? 'def'}`; + // M3 U1 (TsHarvester emits taint `sites` on StatementFacts); pdg:2→3 in the + // #2227 follow-up U1 (every C-family / TS harvester now stamps the call-site + // anchor `SiteRecord.at`, which the resolved-callee-id join reads); pdg:3→4 in + // the #2227 tri-review-2 U4 (the Rust harvester now emits a `kind:'new'` site + // for `struct_expression`, a new worker-output site the join consumes); pdg:4→5 + // in the FU-C call-summary soundness fix (the TS harvester now stamps + // `BindingEntry.formalIndex` on param bindings so the PDG call-summary keys + // return-flow on the enclosing FORMAL position, not the flattened binding + // ordinal — a warm chunk lacking it would route the harvest to its conservative + // empty-summary fallback). A warm chunk built by a worker predating the relevant + // change carries a stale site shape, so the join skips it and + // `BasicBlock.calleeIds` is silently empty (or missing the struct constructor) + // even though `callees` is populated — exactly the #2225-class shape skew this + // version token exists to prevent. Invalidates pdg-mode chunks and their durable + // parsedfile-cache entries; flag-off chunk keys never reach this line and stay + // byte-identical, so non-pdg users pay nothing. Deliberately NOT a SCHEMA_BUMP — + // that gates the whole cache version and would force a full cold re-parse on + // EVERY user (the M1 bump comment above records that cost). + const ns = `pdg:5;maxFn=${opts.maxFunctionLines ?? 'def'}`; return sha256Hex(`${ns}\n${joined}`); }; diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index 9e39a6cc0..2a63d7fb5 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -207,13 +207,36 @@ export interface RepoMeta { * resolved (always present) on every post-#2201 write. */ reachingDefSolver?: string; + /** + * Whether this `--pdg` index recorded the FU-C `CALL_SUMMARY` return-value + * ascent layer (per-callee param→return summary edges). `true` on every + * FU-C+ (v4) write. ABSENT on any pre-FU-C (v3) `--pdg` stamp — that absence + * is what tells `impact`'s PDG mode the index predates CALL_SUMMARY, so it + * surfaces a "no return-value ascent (re-index for CALL_SUMMARY)" note while + * STILL serving the intra slice. CALL_SUMMARY is deliberately NOT a required + * sub-layer for `pdgLayerStatus` to report `'ready'`: a v3 index stays fully + * usable for the intra-procedural statement slice; only the ascent upgrade is + * unavailable. Optional for that back-compat reason. + */ + hasCallSummary?: boolean; }; } /** * Bumped whenever incremental-indexing invariants change incompatibly. + * v2: `BasicBlock.callees` column added (statement-precise inter-procedural + * reach substrate) — an index built before this lacks the column, so a full + * re-analyze is required rather than an incremental top-up. + * v3: `BasicBlock.calleeIds` column added (sound resolved-callee-id parallel + * to `callees`, #2227) — same contract: an index built before this lacks the + * column, so a full re-analyze is forced rather than an incremental top-up. + * v4: `CALL_SUMMARY` relation type added (per-callee RETURN-VALUE ASCENT + * summary edges, PDG FU-C). A pre-v4 `--pdg` index has NO CALL_SUMMARY edges, + * so the engine would silently UNDER-REPORT return-value ascent on an + * incremental top-up; force a full re-analyze instead (same contract as v2/v3). + * This single bump covers the whole FU-C re-index window (and the later FU-B-2). */ -export const INCREMENTAL_SCHEMA_VERSION = 1; +export const INCREMENTAL_SCHEMA_VERSION = 4; export interface IndexedRepo { repoPath: string; diff --git a/gitnexus/test/integration/cfg/cfg-emit.test.ts b/gitnexus/test/integration/cfg/cfg-emit.test.ts index db0540933..66f9a35fc 100644 --- a/gitnexus/test/integration/cfg/cfg-emit.test.ts +++ b/gitnexus/test/integration/cfg/cfg-emit.test.ts @@ -8,6 +8,7 @@ import { emitFileCdg, POST_DOMINATE_DEBUG_ENV, } from '../../../src/core/ingestion/cfg/emit.js'; +import { decodeReachingDefReason } from '../../../src/core/ingestion/cfg/reaching-def-reason-codec.js'; import { getProvider } from '../../../src/core/ingestion/languages/index.js'; import { SupportedLanguages } from '../../../src/config/supported-languages.js'; import type { @@ -220,7 +221,7 @@ describe('U4 — flag-off / empty input emits nothing', () => { }); describe('U4 (#2082 M2) — emitFileReachingDefs', () => { - it('persists deduped (blockPair, binding) edges with reason = plain variable name', () => { + it('persists deduped (blockPair, binding) edges; reason decodes to the variable name + FU-B-2 def/use lines', () => { const cfgs = cfgsOf( `function f(a) { let x = a; @@ -238,10 +239,24 @@ describe('U4 (#2082 M2) — emitFileReachingDefs', () => { expect(e.sourceId).toMatch(/^BasicBlock:src\/rd\.ts:\d+:\d+:\d+$/); expect(e.targetId).toMatch(/^BasicBlock:src\/rd\.ts:\d+:\d+:\d+$/); } - // reason carries the plain source-level name (M0/S1 verdict) - const reasons = new Set(rels.map((e) => e.reason)); - expect(reasons.has('x')).toBe(true); - expect(reasons.has('a')).toBe(true); + // FU-B-2: reason carries the plain source-level name FIRST (M0/S1 verdict) + // plus a versioned def/use-line annotation — decode to recover the name. + const names = new Set(rels.map((e) => decodeReachingDefReason(e.reason).name)); + expect(names.has('x')).toBe(true); + expect(names.has('a')).toBe(true); + // Every emitted edge carries the FU-B-2 line annotation (round-trips to + // finite 1-based def/use source lines) — the substrate the statement-granular + // projection walks. + const decoded = rels.map((e) => decodeReachingDefReason(e.reason)); + for (const d of decoded) { + expect(typeof d.defLine).toBe('number'); + expect(typeof d.useLine).toBe('number'); + } + // The self-edge for `x` (def `let x = a` line 2 → use `x = x + 1` line 3) + // captures the intra-block def@L→use@L' chain the block-pair dedup would lose. + const xSelf = decoded.find((d) => d.name === 'x' && d.defLine !== d.useLine); + expect(xSelf).toMatchObject({ name: 'x' }); + expect(Number(xSelf?.useLine)).toBeGreaterThan(Number(xSelf?.defLine)); }); it('same block pair, two bindings → two distinct edges (id collision-proofing)', () => { @@ -251,7 +266,10 @@ describe('U4 (#2082 M2) — emitFileReachingDefs', () => { const ids = rels.map((e) => e.id); expect(new Set(ids).size).toBe(ids.length); // a and b both flow ENTRY→body: same block pair, distinct edges by binding - const entryToBody = rels.filter((e) => e.reason === 'a' || e.reason === 'b'); + const entryToBody = rels.filter((e) => { + const name = decodeReachingDefReason(e.reason).name; + return name === 'a' || name === 'b'; + }); expect(entryToBody.length).toBeGreaterThanOrEqual(2); }); @@ -267,7 +285,7 @@ describe('U4 (#2082 M2) — emitFileReachingDefs', () => { ); const { graph, rels } = recordingGraph(); const r = emitFileReachingDefs(graph, cfgs); - const xEdges = rels.filter((e) => e.reason === 'x'); + const xEdges = rels.filter((e) => decodeReachingDefReason(e.reason).name === 'x'); expect(xEdges).toHaveLength(1); // self-pair within the single body block expect(r.facts).toBeGreaterThan(rels.length); // facts > deduped edges }); diff --git a/gitnexus/test/integration/cfg/pipeline-pdg.test.ts b/gitnexus/test/integration/cfg/pipeline-pdg.test.ts index 00f4ed8f7..c3759570c 100644 --- a/gitnexus/test/integration/cfg/pipeline-pdg.test.ts +++ b/gitnexus/test/integration/cfg/pipeline-pdg.test.ts @@ -83,6 +83,26 @@ describe('U7 — end-to-end --pdg pipeline', () => { } }, 60000); + // #2227 tri-review-2 R3: the regression guard the stale-parse-cache emptiness + // needed. The bridge unit tests used SYNTHETIC capture maps, so the real + // worker→capture→position-join→BasicBlock.calleeIds path was never exercised + // end-to-end and a stale-cache emptiness shipped undetected. This asserts a + // REAL --pdg run populates calleeIds for the fixture's in-repo call sites + // (read from the in-memory graph; the join runs main-thread in scope-resolution). + it('with --pdg on: populates BasicBlock.calleeIds from the real pipeline (R3)', async () => { + const result = await runPipelineFromRepo(freshRepo(), () => {}, { pdg: true }); + const nonEmptyCalleeIds: string[] = []; + result.graph.forEachNode((n) => { + if (n.label !== 'BasicBlock') return; + const v = n.properties.calleeIds; + if (typeof v === 'string' && v.length > 0) nonEmptyCalleeIds.push(v); + }); + // The fixture has in-repo calls whose resolved ids join into calleeIds; a real + // --pdg run must populate at least one (synthetic-map unit tests cannot catch + // a join/capture/cache regression that empties this column). + expect(nonEmptyCalleeIds.length).toBeGreaterThan(0); + }, 60000); + // M3 (#2083 U4/U7): the taint layer rides the same gate. The fixture's // vuln.ts carries one vulnerable flow (req.body → child_process.exec) and // one sanitized variant (encodeURIComponent before res.send); taint-cases.ts diff --git a/gitnexus/test/integration/cfg/worker-roundtrip.test.ts b/gitnexus/test/integration/cfg/worker-roundtrip.test.ts index 5649f45cb..92f8408da 100644 --- a/gitnexus/test/integration/cfg/worker-roundtrip.test.ts +++ b/gitnexus/test/integration/cfg/worker-roundtrip.test.ts @@ -309,4 +309,26 @@ describe('#2083 M3 U1 — pdg chunk-key namespace version (flag-off keys untouch computeChunkHash(entries, { pdg: true }), ); }); + + it('pdg-mode keys CHANGED from prior namespaces AND pin the current pdg:5 (FU-C BindingEntry.formalIndex)', () => { + // The pdg namespace bumps whenever the worker `cfgSideChannel` SHAPE changes: + // U1 added `SiteRecord.at` (pdg:2→3), U4 added the Rust struct-literal + // `kind:'new'` site (pdg:3→4), and the FU-C call-summary soundness fix added + // `BindingEntry.formalIndex` on param bindings (pdg:4→5) so return-flow keys on + // the enclosing formal position, not the flattened binding ordinal. A stale + // prior shard lacks the new field, so the call-summary harvest would route to + // its conservative empty-summary fallback on a warm cache. Assert prior chunks + // are NOT served, and PIN the current pdg:5 namespace so an accidental revert + // of the token re-introduces the stale-shape bug. + const joined = 'a.ts:h1\nb.ts:h2'; + const keyOf = (token: string) => + createHash('sha256') + .update(Buffer.from(`${token};maxFn=def\n${joined}`)) + .digest('hex'); + const current = computeChunkHash(entries, { pdg: true }); + expect(current).not.toBe(keyOf('pdg:2')); + expect(current).not.toBe(keyOf('pdg:3')); + expect(current).not.toBe(keyOf('pdg:4')); + expect(current).toBe(keyOf('pdg:5')); + }); }); diff --git a/gitnexus/test/integration/impact-pdg-callsummary-degradation.test.ts b/gitnexus/test/integration/impact-pdg-callsummary-degradation.test.ts new file mode 100644 index 000000000..a32facb01 --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-callsummary-degradation.test.ts @@ -0,0 +1,166 @@ +/** + * Integration Test (golden): the NEWEST old-index degradation class — + * a v3 / pre-FU-C `--pdg` index that LACKS the CALL_SUMMARY layer. + * + * The PDG contract is "loud degrade, never silent success." Three old-index + * classes are documented by the new code: + * 1. no CALL_SITE anchors / no calleeIds → impact-pdg-interproc.test.ts + * ("pre-namespace-v4 degrade path": a seed whose blocks carry no calleeIds + * stays intra-only, no cross-function leak). + * 2. no usable PDG layer (CDG/RD absent or partial) → impact-pdg-degradation + * and impact-pdg-id-degradation (the four-state pdgLayerStatus contract). + * 3. CALL_SUMMARY ABSENT (this file). The least-tested case: an index whose + * CDG + REACHING_DEF layers ARE stamped (so the intra slice runs AND the + * inter-procedural descent crosses call boundaries) but whose meta has no + * `hasCallSummary` stamp. The return-value ASCENT silently does nothing — + * so the user MUST be told (a re-index remediation note), never a silent + * "complete" result. + * + * This golden asserts the EXACT degraded envelope (not just non-crash): + * - the result is still mode:'pdg' with pdgResultVersion:1 (the contract + * discriminator); + * - the intra slice is PRESENT (CALL_SUMMARY is NOT a required sub-layer — the + * index is `ready`, pdgLayer is undefined, risk is UNKNOWN, epistemic is the + * real-traversal marker); + * - the note carries the "re-index for CALL_SUMMARY" remediation text — the + * ascent did nothing but the user is TOLD. + * + * Fixture mirrors impact-pdg-interproc.test.ts (fnA calls fnB via calleeIds, a + * downstream RD dependent inside fnB) so the slice genuinely crosses one + * inter-procedural hop — which is the precondition for the CALL_SUMMARY + * remediation note to fire. + */ +import { describe, it, expect, beforeAll, vi } from 'vitest'; +import type { RepoMeta } from '../../src/storage/repo-manager.js'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + // v3 / pre-FU-C index: BOTH caps stamped (⇒ pdgLayerStatus === 'ready', the + // intra slice + inter-procedural descent run) but NO `hasCallSummary` stamp + // (⇒ callSummaryAvailable === false ⇒ return-value ascent is suppressed and + // the remediation note fires). + loadMeta: vi.fn().mockResolvedValue({ + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }, + } as unknown as RepoMeta), + }; +}); + +const F = 'src/callsummary.ts'; + +withTestLbugDB( + 'impact-pdg-callsummary-degradation', + (handle) => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized in afterSetup'); + backend = ext._backend; + }); + + const slice = () => + backend.callTool('impact', { + target: 'fnA', + direction: 'downstream', + mode: 'pdg', + line: 6, + maxDepth: 10, + }); + + describe('CALL_SUMMARY-absent (v3 / pre-FU-C index): the ascent is silent but the user is TOLD', () => { + it('returns the EXACT degraded envelope — mode:pdg, pdgResultVersion:1, intra slice present, risk UNKNOWN', async () => { + const result = await slice(); + // Golden envelope: the index is `ready` (CALL_SUMMARY is NOT a required + // sub-layer), so this is a real traversal result — NOT a pdgLayer + // degradation early-return. The intra slice ran and risk stays UNKNOWN. + expect(result).toMatchObject({ + mode: 'pdg', + pdgResultVersion: 1, + risk: 'UNKNOWN', + epistemic: 'pdg-intra-procedural', + target: { id: 'func:fnA', name: 'fnA' }, + criterionLine: 6, + }); + // CALL_SUMMARY absence does NOT degrade the layer: no pdgLayer marker, + // no probe error, no hard error — the call reached the traversal. + expect(result.pdgLayer).toBeUndefined(); + expect(result.error).toBeUndefined(); + // The intra slice is PRESENT (the index served the statement slice). + expect(result.affectedStatementCount).toBeGreaterThanOrEqual(1); + }); + + it('the note carries the "re-index for CALL_SUMMARY" remediation (ascent did nothing, user is TOLD)', async () => { + const result = await slice(); + // The slice crossed one inter-procedural hop, so the FU-C degradation + // note fires: a caller statement depending on a callee's RETURN value is + // NOT in the slice on a pre-FU-C index — re-index to enable it. + expect(result.note).toMatch(/re-index for CALL_SUMMARY/i); + expect(result.note).toMatch(/CALL_SUMMARY/); + expect(result.note).toMatch(/analyze --pdg/); + // It must NOT read as a confident "complete / no further reach" result. + expect(result.note).not.toMatch(/not yet implemented/i); + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const fn = (id: string, name: string, startLine: number, endLine: number) => + adapter.executePrepared( + `CREATE (n:Function {id: $id, name: $name, filePath: $filePath, startLine: $startLine, endLine: $endLine, isExported: true, content: 'x', description: 'callsummary fixture'})`, + { id, name, filePath: F, startLine, endLine }, + ); + // A BasicBlock carrying the resolved-callee binding the descent keys on. + const block = (id: string, startLine: number, text: string, calleeIds: string) => + adapter.executePrepared( + `CREATE (b:BasicBlock {id: $id, filePath: $filePath, startLine: $startLine, endLine: $startLine, text: $text, callees: '', calleeIds: $calleeIds})`, + { id, filePath: F, startLine, text, calleeIds }, + ); + const edge = (type: 'CDG' | 'REACHING_DEF', src: string, dst: string, reason: string) => + adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: '${type}', confidence: 1.0, reason: $reason, step: 0}]->(b)`, + { src, dst, reason }, + ); + + // Block ids: BasicBlock:::: (fnLine 1-based) + // fnA @0-based[4,8] ⇒ window [5,9]; seed block SA@6 calls fnB. + const SA = `BasicBlock:${F}:5:0:0`; + // fnB @0-based[14,18] ⇒ window [15,19]; SB@16; BD@18 RD-dependent. + const SB = `BasicBlock:${F}:15:0:0`; + const BD = `BasicBlock:${F}:15:0:1`; + + await fn('func:fnA', 'fnA', 4, 8); + await fn('func:fnB', 'fnB', 14, 18); + + await block(SA, 6, 'const a = fnB(x);', 'func:fnB'); + await block(SB, 16, 'const r = compute(y);', ''); + await block(BD, 18, 'return r;', ''); + + // Intra-fnB dependence: SB → BD (line 18 depends on line 16). + await edge('REACHING_DEF', SB, BD, 'r'); + + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'callsummary-repo', + path: '/callsummary/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'callsummary123', + stats: { files: 1, nodes: 5, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as typeof handle & { _backend?: LocalBackend })._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/impact-pdg-degradation.test.ts b/gitnexus/test/integration/impact-pdg-degradation.test.ts new file mode 100644 index 000000000..ee4e41cbb --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-degradation.test.ts @@ -0,0 +1,261 @@ +/** + * Integration Tests: `impact` PDG-mode layer degradation contract (U2 / KTD7) + * + * End-to-end against a REAL LadybugDB, through the full `callTool('impact', …)` + * dispatch. Exercises the four-state PDG-layer presence/degradation check + * (`pdgLayerStatus`) wired into `_impactImpl`'s PDG branch — the check that + * fires after symbol resolution but before traversal so a missing or partial + * `--pdg` layer returns a distinct target-aware guidance note instead of a + * confusing empty blast radius. + * + * The four states (KTD7) are driven by what the (mocked) `loadMeta` returns — + * matching the seeded-DB reality that there is no on-disk `meta.json`: + * - no-layer : meta readable, no `pdg` stamp → run analyze --pdg + * - sub-layer-missing : exactly one cap stamped (CDG xor RD) → names the missing one + * - ready : both caps stamped → falls through to traversal + * - unknown : meta unreadable (null) → inconclusive, via 1 LIMIT 1 probe + * + * The `ready` case asserts the layer check lets the call THROUGH to the real + * traversal, while degraded states return before `_runImpactPDG`. + */ +import { describe, it, expect, beforeAll, beforeEach, vi } from 'vitest'; +import type { RepoMeta } from '../../src/storage/repo-manager.js'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos, loadMeta } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + // Default: meta unreadable (the seeded-DB reality — no on-disk meta.json). + // Individual tests override per state via mockResolvedValueOnce. + loadMeta: vi.fn().mockResolvedValue(null), + }; +}); + +// Minimal seed: one Function symbol (so a `ready` index could resolve it) plus +// a single BasicBlock + CDG edge so the `unknown` state's LIMIT 1 probe finds a +// row (it must STILL stay inconclusive — a present edge cannot disprove an +// edge-free layer / #2188). +const SEED = [ + `CREATE (fn:Function {id: 'func:hot', name: 'hot', filePath: 'src/hot.ts', startLine: 1, endLine: 5, isExported: true, content: 'function hot() {}', description: 'degradation fixture'})`, + `CREATE (b0:BasicBlock {id: 'BasicBlock:src/hot.ts:1:0:0', filePath: 'src/hot.ts', startLine: 2, endLine: 2, text: 'if (x)'})`, + `CREATE (b1:BasicBlock {id: 'BasicBlock:src/hot.ts:1:0:1', filePath: 'src/hot.ts', startLine: 3, endLine: 3, text: 'doThing();'})`, +]; +const SEED_EDGE = `MATCH (a:BasicBlock {id: 'BasicBlock:src/hot.ts:1:0:0'}), (b:BasicBlock {id: 'BasicBlock:src/hot.ts:1:0:1'}) + CREATE (a)-[:CodeRelation {type: 'CDG', confidence: 1.0, reason: 'T', step: 0}]->(b)`; + +const META = (pdg?: RepoMeta['pdg']): RepoMeta => ({ pdg }) as unknown as RepoMeta; + +function expectEmptyPdgParity(result: any): void { + expect(result.mode).toBe('pdg'); + expect(result.direction).toBe('downstream'); + expect(result.impactedCount).toBe(0); + expect(result.risk).toBe('UNKNOWN'); + expect(result.byDepth).toEqual({}); + expect(result.byDepthCounts).toEqual({ 1: 0 }); + expect(result.summary).toEqual({ direct: 0, processes_affected: 0, modules_affected: 0 }); + expect(result.affected_processes).toEqual([]); + expect(result.affected_modules).toEqual([]); +} + +withTestLbugDB( + 'impact-pdg-degradation', + (handle) => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized in afterSetup'); + backend = ext._backend; + }); + + // Reset the loadMeta mock to the default (unreadable) before each test so a + // mockResolvedValueOnce set in one test never leaks into the next. + beforeEach(() => { + vi.mocked(loadMeta).mockReset(); + vi.mocked(loadMeta).mockResolvedValue(null); + }); + + describe('no-layer (meta readable, no pdg stamp)', () => { + it('returns the definitive target-aware "run analyze --pdg" note', async () => { + // Readable meta with no `pdg` key ⇒ the layer was never recorded. + vi.mocked(loadMeta).mockResolvedValueOnce(META(undefined)); + const result = await backend.callTool('impact', { + target: 'hot', + direction: 'downstream', + mode: 'pdg', + }); + + expect(result.mode).toBe('pdg'); + expect(result.pdgLayer).toBe('no-layer'); + expect(result.target).toEqual({ + id: 'func:hot', + name: 'hot', + type: 'Function', + filePath: 'src/hot.ts', + }); + expect(result.note).toMatch(/no PDG layer/i); + expect(result.note).toContain('--pdg'); + // Not a status-unknown note, not a confident LOW. + expect(result.error).toBeUndefined(); + expect(result.note).not.toMatch(/status unknown/i); + expect(result.note).not.toMatch(/not yet implemented/i); + expectEmptyPdgParity(result); + }); + }); + + describe('sub-layer-missing (exactly one cap stamped)', () => { + it('CDG present, RD absent → names REACHING_DEF as missing', async () => { + vi.mocked(loadMeta).mockResolvedValueOnce(META({ maxCdgEdgesPerFunction: 0 } as any)); + const result = await backend.callTool('impact', { + target: 'hot', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.pdgLayer).toBe('sub-layer-missing'); + expect(result.target.filePath).toBe('src/hot.ts'); + expect(result.target.type).toBe('Function'); + expect(result.missingSubLayer).toBe('REACHING_DEF'); + expect(result.note).toMatch(/REACHING_DEF/); + // Partial layer must NOT be reported as complete (no LOW). + expect(result.note).not.toMatch(/not yet implemented/i); + expectEmptyPdgParity(result); + }); + + it('RD present, CDG absent → names CDG as missing', async () => { + vi.mocked(loadMeta).mockResolvedValueOnce( + META({ maxReachingDefEdgesPerFunction: 0 } as any), + ); + const result = await backend.callTool('impact', { + target: 'hot', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.pdgLayer).toBe('sub-layer-missing'); + expect(result.missingSubLayer).toBe('CDG'); + expect(result.note).toMatch(/\bCDG\b/); + expect(result.note).not.toMatch(/not yet implemented/i); + expectEmptyPdgParity(result); + }); + }); + + describe('ready (both caps stamped)', () => { + it('falls THROUGH the layer check to the real traversal (U3 _runImpactPDG)', async () => { + vi.mocked(loadMeta).mockResolvedValueOnce( + META({ maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 } as any), + ); + const result = await backend.callTool('impact', { + target: 'hot', + direction: 'downstream', + mode: 'pdg', + }); + // The layer is complete, so the check did NOT short-circuit: there is no + // degradation note / pdgLayer marker — the call reached the traversal. + expect(result.pdgLayer).toBeUndefined(); + // `hot` has a + // PDG body (blocks B0→B1) but the only dependent (B1) is itself a seed + // block of the symbol, so the intra-procedural downstream reachable set + // is empty — and that is signalled as a real traversal result with the + // distinct "has a body but no dependence" note, NOT the no-body / + // degradation path. The load-bearing U2 fact — + // `ready` does NOT return a degradation note — still holds. + expect(result.mode).toBe('pdg'); + expect(result.error).toBeUndefined(); + expect(result.note).not.toMatch(/not yet implemented/i); + expect(Array.isArray(result.reachableBlocks)).toBe(true); + // Distinct from KTD6 "no PDG body": this symbol HAS a body. + expect(result.epistemic).not.toBe('no-pdg-body'); + }); + + it('a line-seeded slice over callee-less blocks degrades gracefully (real-DB calleesOfBlocks)', async () => { + // The seeded BasicBlocks carry no `callees` data (created without the + // property — the pre-v2 / no-calls reality). A downstream line seed at + // B0 reaches B1 via the CDG edge, so calleesOfBlocks runs over real + // seed+reachable blocks; with no callee data it must yield an empty set + // and degrade to callgraph-equal — no throw, no partial precision. + vi.mocked(loadMeta).mockResolvedValueOnce( + META({ maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 } as any), + ); + const result = await backend.callTool('impact', { + target: 'hot', + direction: 'downstream', + mode: 'pdg', + line: 2, + }); + // The slice resolved (B1 is downstream-dependent on the line-2 seed) and + // the call completed without surfacing a callees-query failure. + expect(result.error).toBeUndefined(); + expect(result.epistemic).toBe('pdg-intra-procedural'); + expect(result.affectedStatementCount).toBeGreaterThanOrEqual(1); + // No CALLS edge in the fixture and no callee data on the blocks, so the + // statement-precise inter-procedural reach is empty — precision is null + // (no reach), never a partial value, and nothing threw. + expect(result.pdgInterprocedural.statementPrecision).toBeNull(); + }); + }); + + describe('unknown (meta unreadable)', () => { + it('returns the inconclusive "status unknown" note via a bounded probe, even with edges present', async () => { + // loadMeta defaults to null (unreadable) via beforeEach. The seeded DB + // DOES carry a CDG edge, but the note must stay inconclusive — a present + // edge cannot prove the layer is complete, and a missing one is + // indistinguishable from an edge-free index (#2188). + const result = await backend.callTool('impact', { + target: 'hot', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.pdgLayer).toBe('unknown'); + expect(result.target.filePath).toBe('src/hot.ts'); + expect(result.target.type).toBe('Function'); + expect(result.note).toMatch(/status unknown/i); + expect(result.note).toContain('--pdg'); + // Inconclusive ≠ definitive no-layer wording. + expect(result.note).not.toMatch(/no PDG layer/i); + expect(result.note).not.toMatch(/not yet implemented/i); + expectEmptyPdgParity(result); + }); + }); + + describe('callgraph mode is unaffected by the PDG-layer probe', () => { + it('mode:callgraph never consults the PDG layer (no degradation note)', async () => { + // Even with meta unreadable, a callgraph impact resolves the symbol and + // returns a real (here: empty-graph) blast radius, never a PDG note. + const result = await backend.callTool('impact', { + target: 'hot', + direction: 'downstream', + mode: 'callgraph', + }); + expect(result.pdgLayer).toBeUndefined(); + // The callgraph path never sets a PDG degradation note. (It may carry + // its own callgraph-flavored notes, but never the PDG-layer wording.) + const note = typeof result.note === 'string' ? result.note : ''; + expect(note).not.toMatch(/status unknown/i); + expect(note).not.toMatch(/no PDG layer/i); + }); + }); + }, + { + seed: [...SEED, SEED_EDGE], + poolAdapter: true, + afterSetup: async (handle) => { + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'degradation-repo', + path: '/degradation/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'deg123', + stats: { files: 1, nodes: 3, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/impact-pdg-e2e.test.ts b/gitnexus/test/integration/impact-pdg-e2e.test.ts new file mode 100644 index 000000000..abd6cd706 --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-e2e.test.ts @@ -0,0 +1,272 @@ +/** + * Integration Tests: real emitter -> persisted PDG rows -> impact(mode:'pdg'). + * + * Seeded PDG traversal tests lock the graph algorithm. This suite locks the + * producer/consumer contract that seeded rows cannot cover: the real analysis + * pipeline must emit BasicBlock ids, source-line metadata, and REACHING_DEF/CDG + * rows in the exact shape consumed by LocalBackend's statement-anchored PDG + * impact traversal and affectedStatements projection. + */ +import { it, expect, beforeAll, vi } from 'vitest'; +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import type { RepoMeta } from '../../src/storage/repo-manager.js'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos } from '../../src/storage/repo-manager.js'; +import { executeParameterized } from '../../src/core/lbug/pool-adapter.js'; +import { runPipelineFromRepo } from '../../src/core/ingestion/pipeline.js'; +import { withTestLbugDB, type IndexedDBHandle } from '../helpers/test-indexed-db.js'; + +const metaByStoragePath = vi.hoisted(() => new Map()); + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + loadMeta: vi.fn().mockImplementation(async (storagePath: string) => { + return metaByStoragePath.get(storagePath) ?? null; + }), + }; +}); + +const FIXTURE = path.join(__dirname, 'cfg', 'fixtures', 'pdg-repo'); +const READY_PDG_META = { + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }, +} as unknown as RepoMeta; + +async function persistFixtureGraph( + pdg: boolean, +): Promise<{ pdgEdges: number; reachingDefEdges: number; cdgEdges: number }> { + const repoDir = fs.mkdtempSync( + path.join(os.tmpdir(), pdg ? 'gn-impact-pdg-' : 'gn-impact-nopdg-'), + ); + try { + fs.cpSync(FIXTURE, repoDir, { recursive: true }); + const pipelineResult = await runPipelineFromRepo(repoDir, () => {}, pdg ? { pdg: true } : {}); + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + const nodes: Array<{ label: 'BasicBlock' | 'Function'; props: Record }> = []; + pipelineResult.graph.forEachNode((n) => { + if (n.label === 'BasicBlock') { + nodes.push({ + label: 'BasicBlock', + props: { + id: n.id, + filePath: n.properties.filePath ?? '', + startLine: n.properties.startLine ?? 0, + endLine: n.properties.endLine ?? 0, + text: n.properties.text ?? '', + }, + }); + } else if (n.label === 'Function') { + nodes.push({ + label: 'Function', + props: { + id: n.id, + name: n.properties.name ?? '', + filePath: n.properties.filePath ?? '', + startLine: n.properties.startLine ?? 0, + endLine: n.properties.endLine ?? 0, + }, + }); + } + }); + + for (const node of nodes) { + const assignments = Object.keys(node.props) + .map((k) => `${k}: $${k}`) + .join(', '); + await adapter.executePrepared( + `CREATE (n:${node.label} {${assignments}})`, + node.props as Record, + ); + } + + let pdgEdges = 0; + let reachingDefEdges = 0; + let cdgEdges = 0; + for (const rel of pipelineResult.graph.iterRelationships()) { + if (rel.type !== 'CDG' && rel.type !== 'REACHING_DEF') continue; + await adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: '${rel.type}', confidence: $confidence, reason: $reason, step: 0}]->(b)`, + { + src: rel.sourceId, + dst: rel.targetId, + confidence: rel.confidence ?? 1.0, + reason: rel.reason ?? '', + }, + ); + pdgEdges++; + if (rel.type === 'REACHING_DEF') reachingDefEdges++; + if (rel.type === 'CDG') cdgEdges++; + } + + return { pdgEdges, reachingDefEdges, cdgEdges }; + } finally { + fs.rmSync(repoDir, { recursive: true, force: true }); + } +} + +function registerSingleRepo(handle: IndexedDBHandle, name: string, repoPath: string): void { + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name, + path: repoPath, + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'impact-pdg-e2e', + stats: { files: 4, nodes: 4, communities: 0, processes: 0 }, + }, + ]); +} + +withTestLbugDB( + 'impact-pdg-e2e', + (handle) => { + let backend: LocalBackend; + let counts: { pdgEdges: number; reachingDefEdges: number; cdgEdges: number }; + + beforeAll(() => { + const ext = handle as typeof handle & { + _backend?: LocalBackend; + _counts?: { pdgEdges: number; reachingDefEdges: number; cdgEdges: number }; + }; + if (!ext._backend || !ext._counts) throw new Error('PDG e2e setup did not finish'); + backend = ext._backend; + counts = ext._counts; + }); + + it('uses real emitted REACHING_DEF and CDG rows to return statement-level PDG impact', async () => { + expect(counts.pdgEdges).toBeGreaterThan(0); + expect(counts.reachingDefEdges).toBeGreaterThan(0); + expect(counts.cdgEdges).toBeGreaterThan(0); + + const result = await backend.callTool('impact', { + target: 'loopFlow', + direction: 'downstream', + mode: 'pdg', + line: 19, + maxDepth: 10, + limit: 50, + }); + + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(result.target.name).toBe('loopFlow'); + expect(result.target.filePath).toBe('guards.ts'); + expect(result.criterionLine).toBe(19); + + const persistedRd = await executeParameterized( + handle.dbPath, + `MATCH (:BasicBlock)-[r:CodeRelation]->(:BasicBlock) + WHERE r.type = 'REACHING_DEF' + RETURN r.type AS type + LIMIT 1`, + {}, + ); + expect(persistedRd.length).toBeGreaterThan(0); + const persistedCdg = await executeParameterized( + handle.dbPath, + `MATCH (:BasicBlock)-[r:CodeRelation]->(:BasicBlock) + WHERE r.type = 'CDG' + RETURN r.type AS type + LIMIT 1`, + {}, + ); + expect(persistedCdg.length).toBeGreaterThan(0); + expect(Array.isArray(result.affectedStatements)).toBe(true); + + const lines = (result.affectedStatements as any[]) + .map((statement) => statement.line) + .sort((a, b) => a - b); + expect(lines).toEqual(expect.arrayContaining([21, 23])); + expect(lines).not.toContain(19); + expect(result.affectedStatementCount).toBe(result.affectedStatements.length); + + const controlResult = await backend.callTool('impact', { + target: 'guarded', + direction: 'downstream', + mode: 'pdg', + line: 9, + maxDepth: 10, + limit: 50, + }); + expect(controlResult.error).toBeUndefined(); + expect(controlResult.mode).toBe('pdg'); + const controlLines = (controlResult.affectedStatements as any[]) + .map((statement) => statement.line) + .sort((a, b) => a - b); + expect(controlLines).toContain(10); + expect(controlLines).not.toContain(9); + }); + }, + { + poolAdapter: true, + timeout: 180_000, + afterSetup: async (handle) => { + metaByStoragePath.set(handle.tmpHandle.dbPath, READY_PDG_META); + const counts = await persistFixtureGraph(true); + if (counts.pdgEdges === 0 || counts.reachingDefEdges === 0 || counts.cdgEdges === 0) { + throw new Error('fixture produced no persisted PDG dependence edges'); + } + registerSingleRepo(handle, 'impact-pdg-e2e', '/impact/pdg/repo'); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + (handle as any)._counts = counts; + }, + }, +); + +withTestLbugDB( + 'impact-pdg-e2e-nopdg', + (handle) => { + let backend: LocalBackend; + + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('no-PDG e2e setup did not finish'); + backend = ext._backend; + }); + + it('returns the no-layer envelope for the same fixture indexed without PDG', async () => { + const result = await backend.callTool('impact', { + target: 'loopFlow', + direction: 'downstream', + mode: 'pdg', + line: 19, + }); + + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(result.pdgLayer).toBe('no-layer'); + expect(result.note).toContain('--pdg'); + expect(result.target).toEqual({ + id: expect.any(String), + name: 'loopFlow', + type: 'Function', + filePath: 'guards.ts', + }); + expect(result.impactedCount).toBe(0); + expect(result.risk).toBe('UNKNOWN'); + expect(result.byDepthCounts).toEqual({ 1: 0 }); + }); + }, + { + poolAdapter: true, + timeout: 180_000, + afterSetup: async (handle) => { + metaByStoragePath.set(handle.tmpHandle.dbPath, {} as RepoMeta); + await persistFixtureGraph(false); + registerSingleRepo(handle, 'impact-pdg-e2e-nopdg', '/impact/no-pdg/repo'); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/impact-pdg-fixtures.test.ts b/gitnexus/test/integration/impact-pdg-fixtures.test.ts new file mode 100644 index 000000000..9b5031c1b --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-fixtures.test.ts @@ -0,0 +1,414 @@ +import { describe, it, expect, afterAll } from 'vitest'; +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { runPipelineFromRepo } from '../../src/core/ingestion/pipeline.js'; +import type { PipelineResult } from '../../src/types/pipeline.js'; + +// U6 — schema-validation + smoke test for the curated impact-PDG accuracy +// fixtures (bench/impact-pdg/fixtures). Two guards: +// +// 1. SCHEMA VALIDATION — every ground-truth.json is well-formed: required keys +// present, a valid `locus`, intra/inter AIS non-overlapping where +// applicable, and the corpus meets the KTD9/F3 minimum floor (>= 3 cases +// per locus stratum, >= 12 measurable). +// +// 2. SMOKE — each fixture analyzes cleanly under {pdg:true} and produces CDG + +// REACHING_DEF edges, AND the criterion function SPECIFICALLY produces both +// (located via the criterion `marker`). A criterion whose function emits +// ZERO PDG edges has unmeasurable ground truth — the smoke test catches it. +// The one intentional no-body case (pdgScoring:"exclude") is the sole +// exemption: its criterion must produce ZERO PDG edges (the KTD6 case). +// +// We annotate from SOURCE SEMANTICS only (KTD9 annotation-circularity guard) — +// this test does NOT derive ground truth from the traversal; it only confirms +// the fixtures are measurable substrate. Reconciling AIS against the live +// traversal is U7's job, not U6's. + +const FIXTURES_DIR = path.join(__dirname, '..', '..', 'bench', 'impact-pdg', 'fixtures'); + +const VALID_LOCI = new Set(['intra', 'inter', 'mixed', 'n/a']); +const VALID_DIRECTIONS = new Set(['downstream', 'upstream']); +const VALID_PROVENANCE = new Set(['manual', 'mutation']); + +interface AisEntry { + symbol: string; + filePath: string; + line?: number; + note?: string; +} +type PdgEdgeKind = 'REACHING_DEF' | 'CDG'; +interface Criterion { + name: string; + filePath: string; + direction: string; + /** + * The 1-based source line of the statement being changed — the seed of the + * statement-anchored PDG slice (`impact({mode:'pdg', line})`, U7 rework). Set + * from SOURCE SEMANTICS (the def/criterion whose change propagates to the + * intra_AIS lines), validated against the live traversal in the harness's + * Step 0. Required for every measurable case; omitted on excluded no-body + * cases (which carry no statement to seed). + */ + line?: number; + marker?: string; + /** + * The PDG edge kinds the criterion function is EXPECTED to produce. A pure + * straight-line data-flow criterion legitimately produces only REACHING_DEF + * (no branches -> no control dependence); a branching/guard criterion + * produces both. The smoke test asserts exactly these are non-zero on the + * criterion, so a pure-dataflow archetype is not forced to carry an + * artificial branch. Omitted on excluded no-body cases. + */ + pdgEdgeKinds?: PdgEdgeKind[]; +} +interface GroundTruth { + schemaVersion: number; + criterion: Criterion; + locus: string; + pdgScoring?: string; + provenance: string; + analyzerVersion: string; + intra_AIS: AisEntry[]; + inter_AIS: AisEntry[]; + rationale: string; +} + +interface FixtureCase { + name: string; + dir: string; + gt: GroundTruth; + excluded: boolean; +} + +function loadFixtures(): FixtureCase[] { + const entries = fs + .readdirSync(FIXTURES_DIR, { withFileTypes: true }) + .filter((d) => d.isDirectory()) + .map((d) => d.name) + .sort(); + return entries.map((name) => { + const dir = path.join(FIXTURES_DIR, name); + const gtPath = path.join(dir, 'ground-truth.json'); + const gt = JSON.parse(fs.readFileSync(gtPath, 'utf8')) as GroundTruth; + return { name, dir, gt, excluded: gt.pdgScoring === 'exclude' }; + }); +} + +const FIXTURES = loadFixtures(); + +function isAisEntry(e: unknown): e is AisEntry { + if (typeof e !== 'object' || e === null) return false; + const o = e as Record; + if (typeof o.symbol !== 'string' || o.symbol.length === 0) return false; + if (typeof o.filePath !== 'string' || o.filePath.length === 0) return false; + if (o.line !== undefined && (typeof o.line !== 'number' || !Number.isInteger(o.line))) { + return false; + } + return true; +} + +/** Stable key for an AIS entry so intra/inter overlap is set-comparable. */ +function aisKey(e: AisEntry): string { + return `${e.symbol}@${e.filePath}#${e.line ?? '-'}`; +} + +const tmpDirs: string[] = []; +function freshRepo(srcDir: string): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-impact-pdg-fx-')); + fs.cpSync(srcDir, dir, { recursive: true }); + tmpDirs.push(dir); + return dir; +} + +interface Counts { + basicBlocks: number; + reachingDefs: number; + cdg: number; +} +function counts(result: PipelineResult): Counts { + let basicBlocks = 0; + result.graph.forEachNode((n) => { + if (n.label === 'BasicBlock') basicBlocks++; + }); + let reachingDefs = 0; + let cdg = 0; + for (const rel of result.graph.iterRelationships()) { + if (rel.type === 'REACHING_DEF') reachingDefs++; + if (rel.type === 'CDG') cdg++; + } + return { basicBlocks, reachingDefs, cdg }; +} + +/** + * BasicBlock id template (cf. emit.ts): + * `BasicBlock::::` + * (filePath may itself contain `:` on some platforms). All blocks of one + * function share the anchor = id minus the trailing `:`. + */ +function functionAnchorOf(blockId: string): string { + return blockId.slice(0, blockId.lastIndexOf(':')); +} + +/** + * Locate the criterion function's block anchor via its `marker` (a substring + * unique to the function body, present in one of its blocks' `text`), then + * count CDG / REACHING_DEF edges SOURCED inside that function. Mirrors the + * `cdgSourcedInHazardFunction` technique in pipeline-pdg.test.ts: attributing + * edges to a specific function, not the whole-fixture aggregate. + */ +function criterionEdgeCounts( + result: PipelineResult, + marker: string, +): { cdg: number; reachingDefs: number; found: boolean } { + let anchor: string | undefined; + const idsByAnchor = new Map>(); + result.graph.forEachNode((n) => { + if (n.label !== 'BasicBlock') return; + const a = functionAnchorOf(n.id); + (idsByAnchor.get(a) ?? idsByAnchor.set(a, new Set()).get(a)!).add(n.id); + const text = (n.properties as { text?: string }).text ?? ''; + if (text.includes(marker)) anchor = a; + }); + if (anchor === undefined) return { cdg: 0, reachingDefs: 0, found: false }; + const blockIds = idsByAnchor.get(anchor) ?? new Set(); + let cdg = 0; + let reachingDefs = 0; + for (const rel of result.graph.iterRelationships()) { + if (!blockIds.has(rel.sourceId)) continue; + if (rel.type === 'CDG') cdg++; + if (rel.type === 'REACHING_DEF') reachingDefs++; + } + return { cdg, reachingDefs, found: true }; +} + +describe('U6 — impact-PDG fixture ground-truth schema', () => { + it('discovers the curated fixtures', () => { + expect(FIXTURES.length).toBeGreaterThan(0); + }); + + for (const fx of FIXTURES) { + describe(fx.name, () => { + const { gt } = fx; + + it('has all required top-level keys', () => { + expect(typeof gt.schemaVersion).toBe('number'); + expect(gt.criterion).toBeTypeOf('object'); + expect(VALID_LOCI.has(gt.locus)).toBe(true); + expect(VALID_PROVENANCE.has(gt.provenance)).toBe(true); + // v1 is manual-annotation-primary (KTD9 — mutation deferred). + expect(gt.provenance).toBe('manual'); + expect(typeof gt.analyzerVersion).toBe('string'); + expect(gt.analyzerVersion.length).toBeGreaterThan(0); + expect(Array.isArray(gt.intra_AIS)).toBe(true); + expect(Array.isArray(gt.inter_AIS)).toBe(true); + expect(typeof gt.rationale).toBe('string'); + // Rationale is what makes manual annotation defensible — non-trivial. + expect(gt.rationale.length).toBeGreaterThan(40); + }); + + it('has a well-formed criterion', () => { + const c = gt.criterion; + expect(typeof c.name).toBe('string'); + expect(c.name.length).toBeGreaterThan(0); + expect(typeof c.filePath).toBe('string'); + expect(c.filePath.startsWith('src/')).toBe(true); + expect(VALID_DIRECTIONS.has(c.direction)).toBe(true); + // The criterion source file actually exists. + expect(fs.existsSync(path.join(fx.dir, c.filePath))).toBe(true); + // Measurable cases carry a marker; excluded no-body cases need none. + if (!fx.excluded) { + expect(typeof c.marker, `${fx.name} needs a criterion.marker`).toBe('string'); + expect(c.marker!.length).toBeGreaterThan(0); + // The marker must appear in the criterion file's source. + const src = fs.readFileSync(path.join(fx.dir, c.filePath), 'utf8'); + expect(src.includes(c.marker!), `marker ${JSON.stringify(c.marker)} in source`).toBe( + true, + ); + // Measurable cases carry a 1-based statement anchor (criterion.line) — + // the seed of the PDG slice (U7 rework). It must be a positive integer + // pointing at an actual source line of the criterion file. + expect(typeof c.line, `${fx.name} needs a 1-based criterion.line`).toBe('number'); + expect(Number.isInteger(c.line!) && c.line! >= 1, `${fx.name} criterion.line >= 1`).toBe( + true, + ); + const lineCount = src.split('\n').length; + expect(c.line! <= lineCount, `${fx.name} criterion.line within file`).toBe(true); + // Measurable cases declare which PDG edge kinds the criterion produces. + expect(Array.isArray(c.pdgEdgeKinds), `${fx.name} needs criterion.pdgEdgeKinds`).toBe( + true, + ); + expect(c.pdgEdgeKinds!.length).toBeGreaterThan(0); + for (const k of c.pdgEdgeKinds!) { + expect(['REACHING_DEF', 'CDG'], `valid pdgEdgeKind ${k}`).toContain(k); + } + } + }); + + it('has well-formed, valid AIS entries', () => { + for (const e of gt.intra_AIS) { + expect(isAisEntry(e), `intra_AIS entry ${JSON.stringify(e)}`).toBe(true); + } + for (const e of gt.inter_AIS) { + expect(isAisEntry(e), `inter_AIS entry ${JSON.stringify(e)}`).toBe(true); + } + }); + + it('has non-overlapping intra/inter AIS', () => { + const intraKeys = new Set(gt.intra_AIS.map(aisKey)); + for (const e of gt.inter_AIS) { + expect(intraKeys.has(aisKey(e)), `inter entry ${aisKey(e)} overlaps intra`).toBe(false); + } + }); + + it('matches its locus to its AIS shape', () => { + if (fx.excluded) { + // Excluded strata: the KTD6 no-body case (locus n/a) AND the resolved-id + // soundness-gate fixture (locus inter) both carry empty AIS — their scoring + // lives on the dedicated id-bridge axis in measure.mjs, not the F1 strata bands, + // so neither intra_AIS nor inter_AIS is the measured quantity. + expect(gt.intra_AIS.length).toBe(0); + expect(gt.inter_AIS.length).toBe(0); + } else if (gt.locus === 'inter') { + // Inter cases: PDG intra-AIS is empty by design; the impact is cross-function. + expect(gt.intra_AIS.length).toBe(0); + expect(gt.inter_AIS.length).toBeGreaterThan(0); + } else if (gt.locus === 'intra') { + // Intra cases: the truly-affected set is within the function. + expect(gt.intra_AIS.length).toBeGreaterThan(0); + expect(gt.inter_AIS.length).toBe(0); + } else { + // Mixed cases: both loci carry genuine impact. (n/a never reaches here — a + // no-body fixture is always pdgScoring:"exclude" and handled above.) + expect(gt.intra_AIS.length).toBeGreaterThan(0); + expect(gt.inter_AIS.length).toBeGreaterThan(0); + } + }); + }); + } + + it('meets the KTD9/F3 minimum corpus floor (>=3 per locus stratum, >=12 measurable)', () => { + const byLocus = new Map(); + let measurable = 0; + for (const fx of FIXTURES) { + if (fx.excluded) continue; + measurable++; + byLocus.set(fx.gt.locus, (byLocus.get(fx.gt.locus) ?? 0) + 1); + } + expect(measurable).toBeGreaterThanOrEqual(12); + for (const locus of ['intra', 'inter', 'mixed']) { + expect(byLocus.get(locus) ?? 0, `>=3 cases for locus ${locus}`).toBeGreaterThanOrEqual(3); + } + }); + + it('exercises BOTH PDG edge kinds across the corpus', () => { + const declared = new Set(); + for (const fx of FIXTURES) { + if (fx.excluded) continue; + for (const k of fx.gt.criterion.pdgEdgeKinds ?? []) declared.add(k); + } + expect(declared.has('REACHING_DEF')).toBe(true); + expect(declared.has('CDG')).toBe(true); + }); + + it('TRIPWIRE: no *.test.ts exists anywhere under bench/impact-pdg/ (cannot inflate npm test)', () => { + // The bench harness (measure.mjs, metrics.mjs, mutation-oracle.mjs) is run + // manually, never by `npm test`. The U2 dynamic-oracle generates instrumented + // mutants in os.tmpdir(), never inside the repo. This tripwire guarantees a + // future probe can never silently drop a `*.test.ts` under bench/impact-pdg/ + // and have it picked up by the default vitest glob — which would inflate the + // suite with a flaky full-pipeline lane the harness is designed to stay out of. + const benchRoot = path.join(__dirname, '..', '..', 'bench', 'impact-pdg'); + const collect = (dir: string): string[] => + fs.readdirSync(dir, { withFileTypes: true }).flatMap((ent) => { + const full = path.join(dir, ent.name); + return ent.isDirectory() ? collect(full) : ent.name.endsWith('.test.ts') ? [full] : []; + }); + const offenders = collect(benchRoot); + expect( + offenders, + `unexpected *.test.ts under bench/impact-pdg/: ${offenders.join(', ')}`, + ).toEqual([]); + }); + + it('has exactly one no-body case (the KTD6 case) and it is excluded', () => { + // The no-body KTD6 case is identified by locus 'n/a' (no CFG body). It is a STRICT + // subset of pdgScoring:"exclude" — the resolved-id soundness-gate fixture is also + // excluded but DOES have a body (locus 'inter'), so we count no-body cases, not all + // excluded cases. + const noBody = FIXTURES.filter((fx) => fx.gt.locus === 'n/a'); + expect(noBody.length).toBe(1); + expect(noBody[0].excluded).toBe(true); + }); +}); + +describe('U6 — impact-PDG fixtures analyze under {pdg:true} with measurable criteria', () => { + afterAll(() => { + for (const d of tmpDirs) fs.rmSync(d, { recursive: true, force: true }); + }); + + for (const fx of FIXTURES) { + if (fx.gt.locus === 'n/a') { + // The intentional no-body case (KTD6): no function bodies, so its criterion must + // produce ZERO PDG edges — a confident zero is the whole point of the exclusion. + it(`${fx.name}: no-body criterion produces ZERO PDG edges (KTD6 exclusion)`, async () => { + const result = await runPipelineFromRepo(freshRepo(fx.dir), () => {}, { pdg: true }); + // The fixture has no function bodies at all, so the whole-repo PDG layer + // is empty too — but the load-bearing claim is the criterion symbol. + const { basicBlocks } = counts(result); + expect(basicBlocks, `${fx.name} no-body fixture should emit no BasicBlocks`).toBe(0); + }, 60000); + continue; + } + + if (fx.excluded) { + // A resolved-id soundness-gate fixture (e.g. intra-overloaded-callee): excluded from + // the F1 strata bands and scored on the dedicated id-bridge axis in measure.mjs, but it + // DOES have function bodies, so it emits a real PDG layer. Assert it analyzes cleanly and + // produces that layer; the F1 criterion-edge checks below intentionally do not apply. + it(`${fx.name}: analyzes under --pdg with a PDG layer (id-bridge soundness gate)`, async () => { + const result = await runPipelineFromRepo(freshRepo(fx.dir), () => {}, { pdg: true }); + const total = counts(result); + expect(total.basicBlocks, `${fx.name} BasicBlock count`).toBeGreaterThan(0); + expect(total.reachingDefs, `${fx.name} REACHING_DEF count`).toBeGreaterThan(0); + }, 60000); + continue; + } + + it(`${fx.name}: analyzes under --pdg; criterion produces its declared PDG edges`, async () => { + const result = await runPipelineFromRepo(freshRepo(fx.dir), () => {}, { pdg: true }); + + // The fixture as a whole produces a PDG layer (BasicBlocks + RD edges). + // CDG is not asserted fixture-wide: a pure straight-line data-flow + // fixture legitimately has zero control dependence. + const total = counts(result); + expect(total.basicBlocks, `${fx.name} BasicBlock count`).toBeGreaterThan(0); + expect(total.reachingDefs, `${fx.name} fixture REACHING_DEF count`).toBeGreaterThan(0); + + // The CRITERION function specifically — located by its marker — must + // produce EXACTLY the edge kinds its ground truth declares, and at least + // one PDG edge overall. A zero-edge criterion has unmeasurable ground + // truth; this is the load-bearing smoke gate (no accidental no-body). + const marker = fx.gt.criterion.marker!; + const crit = criterionEdgeCounts(result, marker); + expect( + crit.found, + `${fx.name} criterion blocks located via marker ${JSON.stringify(marker)}`, + ).toBe(true); + expect( + crit.cdg + crit.reachingDefs, + `${fx.name} criterion '${fx.gt.criterion.name}' produces >=1 PDG edge (unmeasurable if 0)`, + ).toBeGreaterThan(0); + const kinds = fx.gt.criterion.pdgEdgeKinds ?? []; + if (kinds.includes('REACHING_DEF')) { + expect( + crit.reachingDefs, + `${fx.name} criterion declares REACHING_DEF but produced none`, + ).toBeGreaterThan(0); + } + if (kinds.includes('CDG')) { + expect(crit.cdg, `${fx.name} criterion declares CDG but produced none`).toBeGreaterThan(0); + } + }, 60000); + } +}); diff --git a/gitnexus/test/integration/impact-pdg-fullchain-e2e.test.ts b/gitnexus/test/integration/impact-pdg-fullchain-e2e.test.ts new file mode 100644 index 000000000..2dbeddab2 --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-fullchain-e2e.test.ts @@ -0,0 +1,298 @@ +/** + * Integration Test: U4 — FULL-STACK inter-procedural PDG chain. + * + * The sibling suites each cover half of the producer→consumer contract but + * neither chains all four stages with the cross-function hop: + * - impact-pdg-e2e.test.ts runs the REAL emitter but its persist helper drops + * `BasicBlock.calleeIds`, so the U1 inter-procedural descent is never + * exercised against real-emitted ids (an emitter line-encoding change to the + * calleeIds join would not be caught there). + * - impact-pdg-interproc.test.ts exercises the U1 descent but on a HAND-SEEDED + * graph (synthetic block ids + hand-written calleeIds), so an emitter change + * to the basicBlockId template / calleeIdsOfBlock join would not be caught. + * + * This test closes the gap: it runs the REAL pipeline (`runPipelineFromRepo` + * with `pdg:true`) on the pdg-repo fixture — exercising the real emitter's + * `basicBlockId` template AND the `calleeIdsOfBlock` resolved-id join — persists + * the REAL emitted BasicBlock ids + `calleeIds` to a real lbug DB, calls + * `backend.callTool('impact', { mode:'pdg' })` against that DB, and asserts the + * projected `affectedStatements` (a) derive from real-emitted block ids and + * (b) — since U1 landed — CROSS into the called function. + * + * ── The real cross-function shape (taint-cases.ts) ─────────────────────────── + * - `throughCall` (1-based lines 53–57) has a block at startLine 54 whose span + * includes the `const built = decorate(raw);` call on line 55; the REAL emitter + * populates that block's `calleeIds` with the resolved id `Function:taint-cases.ts:decorate`. + * - `decorate` (Function node 0-based [58,60] ⇒ block window [59,61]) carries a + * downstream REACHING_DEF dependent statement at line 60 (`return 'sh -c ' + * + s;`, def→use of the param `s`). + * Seeding `impact(mode:'pdg', target:'throughCall', line:54)` downstream must + * therefore surface line 60 — a statement in a DIFFERENT function reachable ONLY + * via the resolved-callee descent (`interproceduralHops >= 1`). + * + * `loadMeta` is mocked to stamp BOTH PDG caps so `pdgLayerStatus` is `ready`. + */ +import { it, expect, beforeAll, vi } from 'vitest'; +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import type { RepoMeta } from '../../src/storage/repo-manager.js'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos } from '../../src/storage/repo-manager.js'; +import { executeParameterized } from '../../src/core/lbug/pool-adapter.js'; +import { runPipelineFromRepo } from '../../src/core/ingestion/pipeline.js'; +import { withTestLbugDB, type IndexedDBHandle } from '../helpers/test-indexed-db.js'; + +const metaByStoragePath = vi.hoisted(() => new Map()); + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + loadMeta: vi.fn().mockImplementation(async (storagePath: string) => { + return metaByStoragePath.get(storagePath) ?? null; + }), + }; +}); + +const FIXTURE = path.join(__dirname, 'cfg', 'fixtures', 'pdg-repo'); +const READY_PDG_META = { + // hasCallSummary enables the FU-C return-value ascent in the consumer (without + // it the ascent is suppressed and only the descent path is exercised). + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0, hasCallSummary: true }, +} as unknown as RepoMeta; + +/** + * Run the REAL `--pdg` pipeline on the fixture and persist its emitted Function + * + BasicBlock nodes and REACHING_DEF/CDG edges into the active lbug DB. + * + * Unlike impact-pdg-e2e's `persistFixtureGraph`, this persists `calleeIds` (and + * `callees`) on every BasicBlock — the resolved-callee column the U1 descent + * keys on — so the cross-function hop runs against REAL emitter output. + */ +async function persistFixtureGraphWithCallees(): Promise<{ + reachingDefEdges: number; + cdgEdges: number; + calleeIdBearingBlocks: number; + callSummaryEdges: number; +}> { + const repoDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-impact-pdg-fullchain-')); + try { + fs.cpSync(FIXTURE, repoDir, { recursive: true }); + const pipelineResult = await runPipelineFromRepo(repoDir, () => {}, { pdg: true }); + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + let calleeIdBearingBlocks = 0; + const nodes: Array<{ label: 'BasicBlock' | 'Function'; props: Record }> = []; + pipelineResult.graph.forEachNode((n) => { + if (n.label === 'BasicBlock') { + const calleeIds = typeof n.properties.calleeIds === 'string' ? n.properties.calleeIds : ''; + if (calleeIds.length > 0) calleeIdBearingBlocks++; + nodes.push({ + label: 'BasicBlock', + props: { + id: n.id, + filePath: n.properties.filePath ?? '', + startLine: n.properties.startLine ?? 0, + endLine: n.properties.endLine ?? 0, + text: n.properties.text ?? '', + callees: typeof n.properties.callees === 'string' ? n.properties.callees : '', + calleeIds, + }, + }); + } else if (n.label === 'Function') { + nodes.push({ + label: 'Function', + props: { + id: n.id, + name: n.properties.name ?? '', + filePath: n.properties.filePath ?? '', + startLine: n.properties.startLine ?? 0, + endLine: n.properties.endLine ?? 0, + }, + }); + } + }); + + for (const node of nodes) { + const assignments = Object.keys(node.props) + .map((k) => `${k}: $${k}`) + .join(', '); + await adapter.executePrepared( + `CREATE (n:${node.label} {${assignments}})`, + node.props as Record, + ); + } + + let reachingDefEdges = 0; + let cdgEdges = 0; + let callSummaryEdges = 0; + for (const rel of pipelineResult.graph.iterRelationships()) { + // CALL_SUMMARY is a self-loop on the callee's Function/Method node (NOT a + // BasicBlock edge), so persist it separately — the U-C4 return-value ascent + // reads it to re-seed the caller's continuation from the call block. + if (rel.type === 'CALL_SUMMARY') { + await adapter.executePrepared( + `MATCH (a:Function {id: $src}) + CREATE (a)-[:CodeRelation {type: 'CALL_SUMMARY', confidence: $confidence, reason: $reason, step: 0}]->(a)`, + { src: rel.sourceId, confidence: rel.confidence ?? 1.0, reason: rel.reason ?? '' }, + ); + callSummaryEdges++; + continue; + } + if (rel.type !== 'CDG' && rel.type !== 'REACHING_DEF') continue; + await adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: '${rel.type}', confidence: $confidence, reason: $reason, step: 0}]->(b)`, + { + src: rel.sourceId, + dst: rel.targetId, + confidence: rel.confidence ?? 1.0, + reason: rel.reason ?? '', + }, + ); + if (rel.type === 'REACHING_DEF') reachingDefEdges++; + if (rel.type === 'CDG') cdgEdges++; + } + + return { reachingDefEdges, cdgEdges, calleeIdBearingBlocks, callSummaryEdges }; + } finally { + fs.rmSync(repoDir, { recursive: true, force: true }); + } +} + +interface FullChainExt { + _backend?: LocalBackend; + _counts?: { reachingDefEdges: number; cdgEdges: number; calleeIdBearingBlocks: number }; +} + +withTestLbugDB( + 'impact-pdg-fullchain-e2e', + (handle) => { + let backend: LocalBackend; + let counts: NonNullable; + + beforeAll(() => { + const ext = handle as IndexedDBHandle & FullChainExt; + if (!ext._backend || !ext._counts) throw new Error('PDG full-chain e2e setup did not finish'); + backend = ext._backend; + counts = ext._counts; + }); + + it('chains real emitter -> DB -> impact(mode:pdg) and crosses into the called function (U1)', async () => { + // The real emitter populated the resolved-callee column the U1 descent + // keys on, and the data-dependence layer landed. + expect(counts.calleeIdBearingBlocks).toBeGreaterThan(0); + expect(counts.reachingDefEdges).toBeGreaterThan(0); + expect(counts.cdgEdges).toBeGreaterThan(0); + + const result = await backend.callTool('impact', { + target: 'throughCall', + direction: 'downstream', + mode: 'pdg', + line: 54, + maxDepth: 10, + limit: 50, + }); + + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(result.target.name).toBe('throughCall'); + expect(result.target.filePath).toBe('taint-cases.ts'); + expect(result.criterionLine).toBe(54); + + // The persisted block carries the REAL emitted resolved-callee id — this is + // the producer side of the contract the descent consumes. (Read after the + // impact call so the MCP pool adapter for this DB is already open.) + const calleeRows = await executeParameterized( + handle.dbPath, + `MATCH (b:BasicBlock) + WHERE b.filePath = 'taint-cases.ts' AND b.startLine = 54 + RETURN b.id AS id, b.calleeIds AS calleeIds`, + {}, + ); + const calleeCells = (calleeRows as Array>).map((r) => + String(r['calleeIds'] ?? ''), + ); + expect(calleeCells.join(' ')).toContain('Function:taint-cases.ts:decorate'); + expect( + (calleeRows as Array>).map((r) => String(r['id'] ?? '')), + ).toEqual(expect.arrayContaining(['BasicBlock:taint-cases.ts:53:7:2'])); + + // The slice crossed exactly one resolved-callee function hop (throughCall + // -> decorate) — the U1 marker. The hop count is documented in the note + // (the dispatcher folds it into the soundness note rather than a field). + expect(result.note).toMatch(/crosses 1 inter-procedural hop/i); + expect(result.note).toMatch(/resolved call sites/i); + + // The called function `decorate` is surfaced in byDepth — the cross- + // function reach the resolved calleeIds descent unlocked. + const reachedNames = new Set( + Object.values(result.byDepth as Record>) + .flat() + .map((i) => i.name), + ); + expect(reachedNames.has('decorate')).toBe(true); + + // decorate's downstream-dependent statement (line 60, `return 'sh -c ' + + // s;`) lives in a DIFFERENT function — reachable ONLY via the resolved + // calleeIds descent, and the projected line derives from the REAL emitted + // BasicBlock id (anchor `BasicBlock:taint-cases.ts:59:0:2`). + const statements = result.affectedStatements as Array<{ line: number; text: string }>; + const lines = statements.map((statement) => statement.line).sort((a, b) => a - b); + expect(lines).toContain(60); + expect(statements).toEqual( + expect.arrayContaining([ + expect.objectContaining({ line: 60, text: "return 'sh -c ' + s;" }), + ]), + ); + // FU-C wiring (hasCallSummary:true + a REAL persisted CALL_SUMMARY edge): + // line 56 `exec(built);` consumes `built`, decorate's RETURN value. The + // consumer ran calleesWithReturnFlow against the real CALL_SUMMARY edge and + // surfaced the call-result continuation — the meta -> callSummaryAvailable + // -> CALL_SUMMARY-read path that was previously only mock-covered. + expect(lines).toContain(56); + expect(statements).toEqual( + expect.arrayContaining([expect.objectContaining({ line: 56, text: 'exec(built);' })]), + ); + // The seed statement's own line (54) is excluded (seed-minus-reachable). + expect(lines).not.toContain(54); + expect(result.affectedStatementCount).toBe(result.affectedStatements.length); + }); + }, + { + poolAdapter: true, + timeout: 180_000, + afterSetup: async (handle) => { + metaByStoragePath.set(handle.tmpHandle.dbPath, READY_PDG_META); + const counts = await persistFixtureGraphWithCallees(); + if ( + counts.reachingDefEdges === 0 || + counts.calleeIdBearingBlocks === 0 || + counts.callSummaryEdges === 0 + ) { + throw new Error( + 'fixture produced no calleeId-bearing blocks, dependence edges, or CALL_SUMMARY edges', + ); + } + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'impact-pdg-fullchain-e2e', + path: '/impact/pdg/fullchain/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'impact-pdg-fullchain-e2e', + stats: { files: 4, nodes: 4, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + const ext = handle as IndexedDBHandle & FullChainExt; + ext._backend = backend; + ext._counts = counts; + }, + }, +); diff --git a/gitnexus/test/integration/impact-pdg-id-degradation.test.ts b/gitnexus/test/integration/impact-pdg-id-degradation.test.ts new file mode 100644 index 000000000..09ed0f230 --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-id-degradation.test.ts @@ -0,0 +1,233 @@ +/** + * Integration test (U10): the sound resolved-id bridge degrades gracefully on a + * pre-v3 / id-less index — it falls back to the existing leaf-NAME match with no + * crash and no silently-empty proof. End-to-end against a REAL LadybugDB through + * the full `callTool('impact', {mode:'pdg', line})` dispatch. + * + * Covers R3 (graceful degradation) + R7 (the 512-site truncation sentinel keeps a + * capped block callgraph-equal regardless of ids). Two of the three degradation + * shapes are expressible at the integration layer here: + * + * Scenario 1 — empty `calleeIds` cells (the seeded BasicBlocks carry `callees` + * but NO `calleeIds`, so the v3 column reads back empty): `calleeIdsOfBlocks` + * yields an empty id set → the bridge falls back to the NAME path → the reached + * callee whose leaf name is in the slice is proven (callgraph-bridge), exactly + * the pre-feature behavior. Asserted NOT empty / NOT dropped. + * + * Scenario 3 — capped-sentinel block: a slice block whose `callees` carries the + * `*` truncation sentinel (R7) is callee-unknown, so the bridge stays + * callgraph-equal — every reached callee is proven regardless of ids. + * + * Scenario 2 (the `calleeIds` query itself erroring — a truly column-less / older + * binder error) cannot be expressed here: `withTestLbugDB` always materializes the + * current v3 schema, so the column is always present and `RETURN b.calleeIds` + * never raises a binder error. That swallow-on-error → name-fallback path is + * covered at the UNIT layer in + * `test/unit/calltool-dispatch-id-bridge.test.ts` (the "falls back to the name path + * when the calleeIds query errors" case, which mocks `executeParameterized` to + * throw on `RETURN b.calleeIds`). + * + * Module-scoped `let backend` and unconditional asserts, mirroring the sibling PDG + * integration suites (no `if`-branching, no `as any`). + */ +import { it, expect, beforeAll, vi } from 'vitest'; +import type { RepoMeta } from '../../src/storage/repo-manager.js'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos, loadMeta } from '../../src/storage/repo-manager.js'; +import { executeParameterized } from '../../src/core/lbug/pool-adapter.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + loadMeta: vi.fn().mockResolvedValue(null), + }; +}); + +const FILE = 'src/deg.ts'; + +// A Function symbol the dispatch can resolve by name. +const fn = (id: string, name: string, startLine: number, endLine: number): string => + `CREATE (:Function {id: '${id}', name: '${name}', filePath: '${FILE}', startLine: ${startLine}, endLine: ${endLine}, isExported: true, content: 'x', description: 'id-degradation fixture'})`; + +// A BasicBlock with `callees` (leaf names) but WITHOUT `calleeIds` — the property +// is left unset so the present-but-empty v3 column reads back empty (the pre-v3 +// reality: names harvested, resolved ids never captured). +const blkNoIds = (owner: string, idx: number, line: number, callees: string): string => + `CREATE (:BasicBlock {id: 'BasicBlock:${FILE}:${owner}:0:${idx}', filePath: '${FILE}', startLine: ${line}, endLine: ${line}, text: 't', callees: '${callees}'})`; + +const reachingDef = (owner: string, fromIdx: number, toIdx: number): string => + `MATCH (a:BasicBlock {id: 'BasicBlock:${FILE}:${owner}:0:${fromIdx}'}), (b:BasicBlock {id: 'BasicBlock:${FILE}:${owner}:0:${toIdx}'}) CREATE (a)-[:CodeRelation {type: 'REACHING_DEF', confidence: 1.0, reason: 'v', step: 0}]->(b)`; + +const calls = (callerId: string, calleeId: string): string => + `MATCH (c:Function {id: '${callerId}'}), (t:Function {id: '${calleeId}'}) CREATE (c)-[:CodeRelation {type: 'CALLS', confidence: 0.9, reason: 'local-call', step: 0}]->(t)`; + +// ── Scenario 1 graph: `nameCaller` ─────────────────────────────────────────── +// B0 (line 2, seed): calls foo — only foo's call site +// B1 (line 4): calls bar — reachable from B0 via REACHING_DEF +// B2 (line 5): calls baz — NOT reachable from B0 +// slice = {B0, B1}; name-proven = {foo, bar}; baz stays unproven. With empty +// calleeIds the bridge must reproduce this NAME-based verdict. +const SCENARIO_1 = [ + fn('func:nameCaller', 'nameCaller', 0, 6), + fn('func:foo', 'foo', 10, 12), + fn('func:bar', 'bar', 14, 16), + fn('func:baz', 'baz', 18, 20), + blkNoIds('nameCaller', 0, 2, 'foo'), + blkNoIds('nameCaller', 1, 4, 'bar'), + blkNoIds('nameCaller', 2, 5, 'baz'), + reachingDef('nameCaller', 0, 1), + calls('func:nameCaller', 'func:foo'), + calls('func:nameCaller', 'func:bar'), + calls('func:nameCaller', 'func:baz'), +]; + +// ── Scenario 3 graph: `cappedCaller` ───────────────────────────────────────── +// B0 (line 2, seed): callees = '*' (the 512-site truncation sentinel, R7) — the +// block is callee-unknown, so even though calleeIds is empty AND the leaf name +// would not match, the bridge stays callgraph-equal: the reached callee is proven. +const SCENARIO_3 = [ + fn('func:cappedCaller', 'cappedCaller', 30, 36), + fn('func:qux', 'qux', 40, 42), + blkNoIds('cappedCaller', 0, 32, '*'), + calls('func:cappedCaller', 'func:qux'), +]; + +const SEED: string[] = [...SCENARIO_1, ...SCENARIO_3]; + +const READY_META: RepoMeta = { + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }, +} as unknown as RepoMeta; + +let backend: LocalBackend; + +withTestLbugDB( + 'impact-pdg-id-degradation', + (handle) => { + beforeAll(() => { + if (!backend) throw new Error('LocalBackend not initialized in afterSetup'); + }); + + const provenNames = (result: { + pdgInterprocedural?: { statementPreciseByDepth?: Record> }; + }): string[] => + Object.values(result.pdgInterprocedural?.statementPreciseByDepth ?? {}) + .flat() + .map((item) => item.name ?? ''); + + const evidenceById = (result: { + interproceduralByDepth?: Record>; + }): Map => + new Map( + Object.values(result.interproceduralByDepth ?? {}) + .flat() + .map((item) => [item.id ?? '', item.pdgEvidence ?? '']), + ); + + it('precondition: the seeded slice blocks carry empty calleeIds (pre-v3 shape)', async () => { + // Prove the fixture really exercises the degraded path: every BasicBlock has + // an EMPTY calleeIds cell (column present in v3, never populated) so + // `calleeIdsOfBlocks` returns an empty id set and the bridge has no ids to + // prove by — forcing the name fallback / sentinel paths below. + const rows: Array<{ id?: string; calleeIds?: string | null }> = await executeParameterized( + handle.repoId, + `MATCH (b:BasicBlock) RETURN b.id AS id, b.calleeIds AS calleeIds`, + {}, + ); + const calleeIdCells = rows.map((r) => String(r.calleeIds ?? '')); + expect(calleeIdCells.length).toBeGreaterThan(0); + expect(calleeIdCells.every((cell) => cell === '')).toBe(true); + }); + + it('Scenario 1 (R3): empty calleeIds → bridge falls back to the leaf-NAME match', async () => { + vi.mocked(loadMeta).mockResolvedValueOnce(READY_META); + const result = await backend.callTool('impact', { + target: 'nameCaller', + direction: 'downstream', + mode: 'pdg', + line: 2, + }); + + // No crash, no surfaced error — the inter-procedural reach completed. + expect(result.error).toBeUndefined(); + expect(result.epistemic).toBe('pdg-intra-procedural'); + + // The NAME path proves exactly the slice callees: foo (seeded line) and bar + // (dependent block). This is the byte-for-byte pre-feature behavior; an empty + // id set must NOT silently drop the proof. + const proven = provenNames(result); + expect(proven).toContain('foo'); + expect(proven).toContain('bar'); + // baz is reached in the call graph but only from an unreachable block, so it + // stays out of the statement-precise (name-proven) slice. + expect(proven).not.toContain('baz'); + + // The proof is NOT silently empty: there is a statement-precise verdict and a + // non-empty proven set. + expect(result.pdgInterprocedural.statementPreciseImpactedCount).toBeGreaterThanOrEqual(2); + expect(proven.length).toBeGreaterThanOrEqual(2); + + // Per-callee evidence: foo + bar proven via the name bridge, baz unproven — + // the exact verdict the name path (and only the name path, since ids are + // empty) produces. + const byId = evidenceById(result); + expect(byId.get('func:foo')).toBe('callgraph-bridge'); + expect(byId.get('func:bar')).toBe('callgraph-bridge'); + expect(byId.get('func:baz')).toBe('unproven-bridge'); + + // Full call-graph reach is preserved (recall intact) — degradation does not + // shrink the underlying reach, only the proof key. + const fullNames = Object.values(result.pdgInterprocedural?.byDepth ?? {}) + .flat() + .map((item: { name?: string }) => item.name ?? ''); + expect(fullNames).toEqual(expect.arrayContaining(['foo', 'bar', 'baz'])); + }); + + it('Scenario 3 (R7): a capped-sentinel slice block stays callgraph-equal', async () => { + vi.mocked(loadMeta).mockResolvedValueOnce(READY_META); + const result = await backend.callTool('impact', { + target: 'cappedCaller', + direction: 'downstream', + mode: 'pdg', + line: 32, + }); + + expect(result.error).toBeUndefined(); + expect(result.epistemic).toBe('pdg-intra-procedural'); + + // The seed block's call sites are truncated (`*` sentinel) → the callee set is + // incomplete → the bridge keeps the reach callgraph-equal: `qux` is proven + // even though it is neither in calleeIds (empty) nor a literal name match. + const byId = evidenceById(result); + expect(byId.get('func:qux')).toBe('callgraph-bridge'); + + // Callgraph-equal means every reached callee is proven — precision is 1, never + // a partial under-proof, and nothing was dropped. + expect(result.pdgInterprocedural.statementPrecision).toBe(1); + const proven = provenNames(result); + expect(proven).toContain('qux'); + }); + }, + { + seed: SEED, + poolAdapter: true, + afterSetup: async (handle) => { + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'id-degradation-repo', + path: '/id-degradation/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'iddeg123', + stats: { files: 1, nodes: 6, communities: 0, processes: 0 }, + }, + ]); + backend = new LocalBackend(); + await backend.init(); + }, + }, +); diff --git a/gitnexus/test/integration/impact-pdg-interproc.test.ts b/gitnexus/test/integration/impact-pdg-interproc.test.ts new file mode 100644 index 000000000..b0261a02a --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-interproc.test.ts @@ -0,0 +1,214 @@ +/** + * Integration Test: U1 — bounded INTER-PROCEDURAL PDG forward slice. + * + * The intra-procedural slice (CDG + REACHING_DEF) stays inside one function. + * U1 descends through resolved call sites (`BasicBlock.calleeIds`) so the + * statement slice crosses function boundaries — HRB context-insensitive forward + * closure (Joern's shipped approach), DOWNSTREAM only, bounded by a function-hop + * depth budget + a shared `visited` set. + * + * ── Fixture graph (hand-seeded, no parser; controlled line numbers) ────────── + * Two files mirror the resolved-callee binding the descent keys on. + * - `fnA` at 0-based [4,8] (window [5,9]); seed block SA@6 carries + * `calleeIds = 'func:fnB'` (fnA calls fnB on the seeded line 6). + * - `fnB` at 0-based [14,18] (window [15,19]); seed block SB@16, with a + * downstream REACHING_DEF dependent BD@18 ('return r;') inside fnB. + * - `fnC` at 0-based [24,28]; SB@16 carries `calleeIds = 'func:fnC'`, so the + * descent reaches a SECOND hop (fnB calls fnC). fnC has dependent CD@27. + * Seeding `impact(mode:'pdg', line:6, target:'fnA')` downstream must surface + * fnB's dependent statement (line 18) AND fnC's (line 27) in affectedStatements + * — neither is reachable by the intra slice (they live in other functions). + * + * A control fixture `fnNoCall` carries NO `calleeIds` (the pre-namespace-v4 + * degrade path): its slice must stay intra-only (no cross-function leak). + * + * `loadMeta` is mocked to stamp BOTH caps so `pdgLayerStatus` returns `ready`. + */ +import { describe, it, expect, beforeAll, vi } from 'vitest'; +import type { RepoMeta } from '../../src/storage/repo-manager.js'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + // Both caps stamped ⇒ pdgLayerStatus === 'ready' ⇒ traversal + descent run. + loadMeta: vi.fn().mockResolvedValue({ + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }, + } as unknown as RepoMeta), + }; +}); + +const F = 'src/interproc.ts'; + +withTestLbugDB( + 'impact-pdg-interproc', + (handle) => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized in afterSetup'); + backend = ext._backend; + }); + + const slice = () => + backend.callTool('impact', { + target: 'fnA', + direction: 'downstream', + mode: 'pdg', + line: 6, + maxDepth: 10, + }); + + describe('cross-function forward closure (U1)', () => { + it('reaches a dependent statement in a CALLED function (fnA -> fnB) in affectedStatements', async () => { + const result = await slice(); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(result.target.id).toBe('func:fnA'); + expect(result.criterionLine).toBe(6); + const lines = (result.affectedStatements as Array<{ line: number }>) + .map((s) => s.line) + .sort((a, b) => a - b); + // fnB's downstream-dependent statement (line 18) is cross-function — only + // the inter-procedural descent surfaces it. + expect(lines).toContain(18); + }); + + it('descends a SECOND hop (fnB -> fnC) within the depth budget', async () => { + const result = await slice(); + const lines = (result.affectedStatements as Array<{ line: number }>) + .map((s) => s.line) + .sort((a, b) => a - b); + // fnC is reached only via fnB's call site (two function hops from fnA). + expect(lines).toContain(27); + }); + + it('projects the called functions to owning symbols in byDepth (single collapsed bucket)', async () => { + const result = await slice(); + // byDepth stays the single collapsed bucket (block-hops are not call-hops); + // the cross-function reach DEEPENS the statement slice, not the bucket count. + expect(Object.keys(result.byDepth)).toEqual(['1']); + const names = new Set( + Object.values(result.byDepth as Record>) + .flat() + .map((i) => i.name), + ); + expect(names.has('fnB')).toBe(true); + expect(names.has('fnC')).toBe(true); + }); + + it('documents the 4 soundness caveats in the note when the slice crosses a hop', async () => { + const result = await slice(); + expect(result.note).toMatch(/return-value ascent/i); + expect(result.note).toMatch(/context-insensitive/i); + expect(result.note).toMatch(/alias model/i); + expect(result.note).toMatch(/resolver/i); + }); + + it('risk stays UNKNOWN (never a confident LOW) and the result is consumer-shaped', async () => { + const result = await slice(); + expect(result.risk).toBe('UNKNOWN'); + expect(result.risk).not.toBe('LOW'); + expect(result.affected_processes).toEqual([]); + expect(result.affected_modules).toEqual([]); + }); + }); + + describe('pre-namespace-v4 degrade path (no calleeIds)', () => { + it('a seed whose blocks carry no calleeIds stays intra-only (no cross-function leak)', async () => { + const result = await backend.callTool('impact', { + target: 'fnNoCall', + direction: 'downstream', + mode: 'pdg', + line: 36, + maxDepth: 10, + }); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + const lines = (result.affectedStatements as Array<{ line: number }>) + .map((s) => s.line) + .sort((a, b) => a - b); + // The intra dependent (line 38) is present; no foreign-function line leaks. + expect(lines).toContain(38); + expect(lines).not.toContain(18); + expect(lines).not.toContain(27); + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const fn = (id: string, name: string, startLine: number, endLine: number) => + adapter.executePrepared( + `CREATE (n:Function {id: $id, name: $name, filePath: $filePath, startLine: $startLine, endLine: $endLine, isExported: true, content: 'x', description: 'interproc fixture'})`, + { id, name, filePath: F, startLine, endLine }, + ); + // A BasicBlock that carries the resolved-callee binding the descent keys on. + const block = (id: string, startLine: number, text: string, calleeIds: string) => + adapter.executePrepared( + `CREATE (b:BasicBlock {id: $id, filePath: $filePath, startLine: $startLine, endLine: $startLine, text: $text, callees: '', calleeIds: $calleeIds})`, + { id, filePath: F, startLine, text, calleeIds }, + ); + const edge = (type: 'CDG' | 'REACHING_DEF', src: string, dst: string, reason: string) => + adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: '${type}', confidence: 1.0, reason: $reason, step: 0}]->(b)`, + { src, dst, reason }, + ); + + // Block ids: BasicBlock:::: (fnLine 1-based) + // fnA @0-based[4,8] ⇒ window [5,9]; seed block SA@6 calls fnB. + const SA = `BasicBlock:${F}:5:0:0`; + // fnB @0-based[14,18] ⇒ window [15,19]; SB@16 calls fnC; BD@18 RD-dependent. + const SB = `BasicBlock:${F}:15:0:0`; + const BD = `BasicBlock:${F}:15:0:1`; + // fnC @0-based[24,28] ⇒ window [25,29]; SC@26; CD@27 RD-dependent. + const SC = `BasicBlock:${F}:25:0:0`; + const CD = `BasicBlock:${F}:25:0:1`; + // fnNoCall @0-based[34,38] ⇒ window [35,39]; NS@36 (no calleeIds); ND@38. + const NS = `BasicBlock:${F}:35:0:0`; + const ND = `BasicBlock:${F}:35:0:1`; + + await fn('func:fnA', 'fnA', 4, 8); + await fn('func:fnB', 'fnB', 14, 18); + await fn('func:fnC', 'fnC', 24, 28); + await fn('func:fnNoCall', 'fnNoCall', 34, 38); + + await block(SA, 6, 'const a = fnB(x);', 'func:fnB'); + await block(SB, 16, 'const r = fnC(y);', 'func:fnC'); + await block(BD, 18, 'return r;', ''); + await block(SC, 26, 'const c = work(z);', ''); + await block(CD, 27, 'return c;', ''); + await block(NS, 36, 'const n = local(p);', ''); // no calleeIds → intra-only + await block(ND, 38, 'return n;', ''); + + // Intra-fnB dependence: SB → BD (line 18 depends on line 16). + await edge('REACHING_DEF', SB, BD, 'r'); + // Intra-fnC dependence: SC → CD (line 27 depends on line 26). + await edge('REACHING_DEF', SC, CD, 'c'); + // Intra-fnNoCall dependence: NS → ND (line 38 depends on line 36). + await edge('REACHING_DEF', NS, ND, 'n'); + + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'interproc-repo', + path: '/interproc/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'interproc123', + stats: { files: 1, nodes: 11, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as typeof handle & { _backend?: LocalBackend })._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/impact-pdg-shape.test.ts b/gitnexus/test/integration/impact-pdg-shape.test.ts new file mode 100644 index 000000000..0458ab03d --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-shape.test.ts @@ -0,0 +1,608 @@ +/** + * Integration Tests: `impact` PDG-mode RESULT SHAPE + consumer-safety (U4 / KTD8) + * + * This suite guards the **standing interchangeability contract** (KTD8): a + * `mode:'pdg'` result must be structurally substitutable for the call-graph + * result for EVERY consumer (CLI `formatImpactResult`, group + * `collectImpactSymbolUids`/`mergeRisk`, `impactByUid`). These are not one-time + * checks — they protect a permanent contract. If a future change drops `target.id`, + * un-collapses `byDepth`, or mints a non-`UNKNOWN` risk, a consumer misrenders and + * one of these tests must go red. + * + * It also exercises the **net-new block→owning-symbol resolver** (the reverse of + * `resolveBlockAnchor`, no precedent) including the two non-happy paths the + * Feasibility review surfaced: + * - same-line, different-name functions → ambiguous-projection (report ALL, + * never silently pick one — there is no `startColumn` to disambiguate), and + * - a reachable block that owns no symbol (top-level / free statement) → + * reported under its file as `unresolved`, NEVER silently dropped (R9). + * + * ── Fixture graph (hand-seeded, no parser; controlled line numbers) ────────── + * One file `src/flow.ts`. + * - `target` fn at 0-based [10,10] ⇒ anchor window [11,11], seed block S@11. + * - `up` fn at [4,4] ⇒ block P@6 (RD def + CDG controller into S). + * - `down` fn at [19,21] ⇒ blocks D1@21, D2@22 (RD uses, downstream of S). + * - `ctl` fn at [29,31] ⇒ blocks K1@31, K2@32 (CDG dependents of S). + * - `dupA` AND `dupB`, BOTH at 0-based [40,42] (SAME (filePath,startLine)) — + * a CDG-reachable block T@41 maps to BOTH (ambiguous-projection). + * - `FlowThing.constructor` at 0-based [55,55] owns CT@56 (constructor projection). + * - a free/top-level block U@99 owned by NO symbol (downstream of K2) — the + * `unresolved` shadow path. + * + * Downstream from S: RD → {D1,D2,CT}; CDG → {K1,K2} → T(@41) → U(@99 top-level). + * + * `loadMeta` is mocked to stamp BOTH caps so `pdgLayerStatus` returns `ready`. + */ +import { describe, it, expect, beforeAll, vi } from 'vitest'; +import type { RepoMeta } from '../../src/storage/repo-manager.js'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; +import { collectImpactSymbolUids, mergeRisk } from '../../src/core/group/cross-impact.js'; +import type { CrossRepoImpact } from '../../src/core/group/types.js'; + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + // Both caps stamped ⇒ pdgLayerStatus === 'ready' ⇒ traversal + projection run. + loadMeta: vi.fn().mockResolvedValue({ + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }, + } as unknown as RepoMeta), + }; +}); + +const F = 'src/flow.ts'; +// Block ids: BasicBlock:::: (fnLine 1-based) +const S = `BasicBlock:${F}:11:0:0`; // target@[10,10] +const P = `BasicBlock:${F}:5:0:0`; // up@[4,4] +const D1 = `BasicBlock:${F}:20:0:0`; // down@[19,21] +const D2 = `BasicBlock:${F}:20:0:1`; // down@[19,21] +const K1 = `BasicBlock:${F}:30:0:0`; // ctl@[29,31] +const K2 = `BasicBlock:${F}:30:0:1`; // ctl@[29,31] +const T = `BasicBlock:${F}:41:0:0`; // dupA AND dupB BOTH @[40,42] → ambiguous +const CT = `BasicBlock:${F}:56:0:0`; // Constructor FlowThing.constructor@[55,55] +const U = `BasicBlock:${F}:99:0:0`; // top-level / no owning symbol → unresolved + +withTestLbugDB( + 'impact-pdg-shape', + (handle) => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized in afterSetup'); + backend = ext._backend; + }); + + // Deep enough to traverse the full CDG chain S→K1→K2→T(@dup)→U(top-level), + // so the ambiguous-projection and unresolved shadow-path blocks are reached. + const downstream = () => + backend.callTool('impact', { + target: 'target', + direction: 'downstream', + mode: 'pdg', + maxDepth: 10, + }); + + // ── The net-new block → owning-symbol resolver ──────────────────────────── + describe('block → owning-symbol projection (KTD8 net-new resolver)', () => { + it('maps a known reachable block to its owning function (0-based offset correct)', async () => { + const result = await downstream(); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + // D1/D2 (fnLine 20 ⇒ symbol startLine 19) own `down`; K1/K2 (fnLine 30 ⇒ + // startLine 29) own `ctl`. The resolver must surface BOTH owning fns. + const items = Object.values(result.byDepth as Record).flat(); + const names = new Set(items.map((i: any) => i.name)); + expect(names.has('down')).toBe(true); + expect(names.has('ctl')).toBe(true); + // And the resolved owning symbols carry real UIDs (not null). + const down = items.find((i: any) => i.name === 'down'); + expect(down.id).toBe('func:down'); + expect(down.filePath).toBe(F); + }); + + it('maps a reachable constructor block to its owning Constructor symbol', async () => { + const result = await downstream(); + const items = Object.values(result.byDepth as Record).flat(); + const ctor = items.find((i: any) => i.id === 'ctor:FlowThing'); + expect(ctor).toBeDefined(); + expect(ctor.name).toBe('FlowThing.constructor'); + expect(ctor.type).toBe('Constructor'); + expect(ctor.filePath).toBe(F); + }); + + it('a reachable block owning NO symbol is reported as unresolved, never dropped (R9 shadow path)', async () => { + const result = await downstream(); + const items = Object.values(result.byDepth as Record).flat(); + // U@99 has no owning Function/Method/Constructor → an explicit unresolved entry. + const unresolved = items.filter((i: any) => i.id === null || i.type === 'unresolved'); + expect(unresolved.length).toBeGreaterThanOrEqual(1); + expect(unresolved[0].filePath).toBe(F); + // It is surfaced (recall preserved), and the top-level count is exposed. + expect(result.unresolvedBlockCount).toBeGreaterThanOrEqual(1); + // But an unresolved block contributes NO symbol UID (no false attribution). + expect(unresolved.every((i: any) => i.id === null)).toBe(true); + }); + + it('two functions sharing (filePath, startLine) project to BOTH, never a silent pick (Feasibility Finding 1)', async () => { + const result = await downstream(); + const items = Object.values(result.byDepth as Record).flat(); + // T@41 ⇒ startLine 40 matches BOTH func:dupA and func:dupB (they share the + // NAME 'dupTarget' and the same start line). The resolver must report + // BOTH (ambiguous-projection), each flagged, never one of them. Identity + // is the UID, not the shared name. + const dupItems = items.filter((i: any) => i.id === 'func:dupA' || i.id === 'func:dupB'); + const dupIds = new Set(dupItems.map((i: any) => i.id)); + expect(dupIds.has('func:dupA')).toBe(true); + expect(dupIds.has('func:dupB')).toBe(true); + expect(dupItems.every((i: any) => i.ambiguous === true)).toBe(true); + // Both share the same projected name (the schema can't disambiguate). + expect(dupItems.every((i: any) => i.name === 'dupTarget')).toBe(true); + expect(result.ambiguousProjectionCount).toBeGreaterThanOrEqual(1); + // The PDG note must call out the ambiguity, not hide it. + expect(result.note).toMatch(/ambiguous|same-line/i); + }); + }); + + // ── KTD8 result-shape parity matrix (vs the call-graph result) ──────────── + describe('result-shape parity (KTD8 standing interchangeability contract)', () => { + it('populates target.id / target.filePath with call-graph-compatible shape', async () => { + const result = await downstream(); + expect(result.target).toBeDefined(); + expect(result.target.id).toBe('func:target'); + expect(result.target.name).toBe('target'); + expect(result.target.filePath).toBe(F); + // `type` present like the callgraph target (consumers may read it). + expect(typeof result.target.type).toBe('string'); + }); + + it('populates byDepth (single collapsed bucket) and byDepthCounts in callgraph shape', async () => { + const result = await downstream(); + // byDepth is a { [depth]: item[] } map, collapsed to exactly ONE bucket + // — block-hops are not call-hops, so there is no multi-depth fan. + const depths = Object.keys(result.byDepth); + expect(depths).toEqual(['1']); + expect(Array.isArray(result.byDepth['1'])).toBe(true); + // Each item carries the call-graph item fields consumers iterate on. + for (const it of result.byDepth['1']) { + expect(it).toHaveProperty('id'); + expect(it).toHaveProperty('name'); + expect(it).toHaveProperty('filePath'); + expect(it).toHaveProperty('processes'); // shape-stable like callgraph + } + // byDepthCounts mirrors the bucket. + expect(result.byDepthCounts['1']).toBe(result.byDepth['1'].length); + }); + + it('affected_processes / affected_modules are empty arrays (consumers coalesce []) ', async () => { + const result = await downstream(); + expect(result.affected_processes).toEqual([]); + expect(result.affected_modules).toEqual([]); + expect(result.summary.processes_affected).toBe(0); + expect(result.summary.modules_affected).toBe(0); + }); + + it('epistemic/note is PDG-specific, NOT the callgraph DI/dynamic-dispatch copy', async () => { + const result = await downstream(); + // PDG marker, not the callgraph 'lower-bound'/'exact'. + expect(result.epistemic).toBe('pdg-intra-procedural'); + // Note frames the intra-procedural caveat, never the DI/interface text. + expect(result.note).toMatch(/intra-procedural|dependence/i); + expect(result.note).not.toMatch(/DI container|dynamic dispatch/i); + }); + + it("risk is the existing 'UNKNOWN' sentinel, not a minted PDG label", async () => { + const result = await downstream(); + expect(result.risk).toBe('UNKNOWN'); + // impactedCount = distinct owning SYMBOLS (down, ctl, dupA, dupB, ctor) — + // the meaningful unit; unresolved blocks do not inflate it. + expect(result.impactedCount).toBe(5); + // blockCount is the raw reachable-block count, retained separately. + expect(result.blockCount).toBeGreaterThanOrEqual(result.impactedCount); + }); + }); + + // ── Statement-mode result shape (criterionLine + slice + KTD8 parity) ───── + describe('statement-mode result carries the slice fields AND the KTD8 parity fields', () => { + it('a line-seeded result has criterionLine/affectedStatements/affectedStatementCount', async () => { + const result = await backend.callTool('impact', { + target: 'accum', + direction: 'downstream', + mode: 'pdg', + line: 72, + }); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + // Statement-mode-specific fields. + expect(result.criterionLine).toBe(72); + expect(Array.isArray(result.affectedStatements)).toBe(true); + expect(result.affectedStatementCount).toBe(result.affectedStatements.length); + expect(result.affectedStatementCount).toBe(2); + for (const s of result.affectedStatements) { + expect(s).toHaveProperty('line'); + expect(s).toHaveProperty('filePath'); + expect(s).toHaveProperty('text'); + } + // The slice statements are the downstream-dependent ones (lines 10, 12). + const lines = (result.affectedStatements as any[]).map((s) => s.line).sort((a, b) => a - b); + expect(lines).toEqual([74, 76]); + }); + + it('the statement-mode result ALSO carries the KTD8 parity fields (byDepth/target/risk/empty processes-modules)', async () => { + const result = await backend.callTool('impact', { + target: 'accum', + direction: 'downstream', + mode: 'pdg', + line: 72, + }); + // byDepth is the single collapsed bucket (block-hops ≠ call-hops). + expect(Object.keys(result.byDepth)).toEqual(['1']); + expect(result.byDepthCounts['1']).toBe(result.byDepth['1'].length); + // target carries the call-graph-compatible shape. + expect(result.target.id).toBe('func:accum'); + expect(result.target.name).toBe('accum'); + expect(result.target.filePath).toBe(F); + expect(typeof result.target.type).toBe('string'); + // risk is the UNKNOWN sentinel (never a confident LOW). + expect(result.risk).toBe('UNKNOWN'); + expect(result.risk).not.toBe('LOW'); + // Empty processes/modules — consumers coalesce []. + expect(result.affected_processes).toEqual([]); + expect(result.affected_modules).toEqual([]); + expect(result.summary.processes_affected).toBe(0); + expect(result.summary.modules_affected).toBe(0); + }); + }); + + // ── Consumer safety: group cross-impact treats the PDG result correctly ─── + describe('group consumer safety (cross-impact.ts)', () => { + it('collectImpactSymbolUids collects the PDG owning-symbol UIDs (non-zero)', async () => { + const result = await downstream(); + const { uids, targetFilePath } = collectImpactSymbolUids(result, undefined); + // The target plus every resolved owning symbol's UID is collected; the + // null unresolved entry contributes nothing (no crash, no '' UID). + expect(uids.length).toBeGreaterThan(0); + expect(uids).toContain('func:target'); // from target.id + expect(uids).toContain('func:down'); + expect(uids).toContain('func:ctl'); + expect(uids).toContain('ctor:FlowThing'); + expect(uids).not.toContain('null'); + expect(uids).not.toContain(''); + expect(targetFilePath).toBe(F); + }); + + it("mergeRisk does NOT render a confident LOW from a PDG 'UNKNOWN' risk (R7 false-LOW trap)", async () => { + const result = await downstream(); + const localRisk = String(result.risk); + expect(localRisk).toBe('UNKNOWN'); + // No cross-repo hits → mergeRisk returns localRisk verbatim: 'UNKNOWN', + // NEVER coerced to a confident 'LOW' (the false-safe this guards). + expect(mergeRisk(localRisk, [])).toBe('UNKNOWN'); + // With a cross-repo hit, 'UNKNOWN' bumps UP to 'MEDIUM' (never down to LOW). + const cross = [{ contract: { confidence: 0.5 } }] as unknown as CrossRepoImpact[]; + expect(mergeRisk(localRisk, cross)).toBe('MEDIUM'); + }); + }); + + // ── KTD5 ambiguous trap: pdg+ambiguous never runs interprocedural fan-out ── + describe('KTD5 ambiguous target never invokes the interprocedural BFS', () => { + it("mode:'pdg' on an ambiguous target returns candidates, never calls _runImpactBFS", async () => { + // Spy on the private interprocedural BFS; ambiguous PDG has no single + // resolved symbol, so it must not run the composed symbol-reach pass. + // `dupTarget` collides across dupA/dupB by NAME. + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + try { + const result = await backend.callTool('impact', { + target: 'dupTarget', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.status).toBe('ambiguous'); + expect(result.mode).toBe('pdg'); + // No interprocedural symbol-reach pass ran without a resolved target. + expect(bfsSpy).not.toHaveBeenCalled(); + // And it surfaces the candidate list (no silent zero blast radius). + expect(Array.isArray(result.candidates)).toBe(true); + expect(result.candidates.length).toBeGreaterThanOrEqual(2); + // The response never names pdg_query (toolName union widened to impact). + expect(JSON.stringify(result)).not.toMatch(/pdg_query/); + } finally { + bfsSpy.mockRestore(); + } + }); + }); + + // ── FIX 1 keystone: the PDG seed anchors on the ALREADY-RESOLVED symbol ─── + // The seed must NOT be re-resolved by bare `sym.name` inside `_runImpactPDG` + // (that would re-ambiguate a file_path/uid-disambiguated name, or anchor on a + // DIFFERENT same-name symbol → wrong-symbol blast radius). With two functions + // named `sameName` in different files, disambiguating by file_path/target_uid + // must produce the CORRECT local PDG blast radius before the composed + // interprocedural `_runImpactBFS` pass runs for the resolved symbol. + describe('seed anchors on the resolved (disambiguated) symbol, not a name re-resolution', () => { + it('file_path disambiguation reaches the right file’s downstream owner (not the other same-name fn)', async () => { + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + try { + const result = await backend.callTool('impact', { + target: 'sameName', + file_path: 'src/b.ts', + direction: 'downstream', + mode: 'pdg', + }); + // Resolved cleanly (NOT the ambiguous early payload) and it is a PDG result. + expect(result.status).not.toBe('ambiguous'); + expect(result.mode).toBe('pdg'); + expect(result.error).toBeUndefined(); + // The target is the B-file `sameName`, and the blast radius reflects B's + // downstream owner `onlyB` — NEVER A's `onlyA` (the wrong-file anchor). + expect(result.target.id).toBe('func:sameB'); + expect(result.target.filePath).toBe('src/b.ts'); + const names = new Set( + Object.values(result.byDepth as Record) + .flat() + .map((i: any) => i.name), + ); + expect(names.has('onlyB')).toBe(true); + expect(names.has('onlyA')).toBe(false); + // Unified PDG now composes interprocedural symbol reach after the local + // PDG slice anchors on the resolved symbol. + expect(bfsSpy).toHaveBeenCalledTimes(1); + } finally { + bfsSpy.mockRestore(); + } + }); + + it('target_uid disambiguation anchors the seed on THAT uid (not a re-ambiguation)', async () => { + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + try { + const result = await backend.callTool('impact', { + target: 'sameName', + target_uid: 'func:sameA', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.status).not.toBe('ambiguous'); + expect(result.mode).toBe('pdg'); + expect(result.target.id).toBe('func:sameA'); + expect(result.target.filePath).toBe('src/a.ts'); + const names = new Set( + Object.values(result.byDepth as Record) + .flat() + .map((i: any) => i.name), + ); + expect(names.has('onlyA')).toBe(true); + expect(names.has('onlyB')).toBe(false); + expect(bfsSpy).toHaveBeenCalledTimes(1); + } finally { + bfsSpy.mockRestore(); + } + }); + }); + + // ── fnFileOf Windows-path coverage (split-from-right) ───────────────────── + // The block→owning-symbol projector (`projectBlocksToSymbols`) recovers each + // block's file path via `fnFileOf`, which must split a `BasicBlock` id FROM + // THE RIGHT so a Windows drive-letter ':' inside the path is not mistaken for + // a segment delimiter. Exercised behaviorally (fnFileOf is module-scope, not + // exported) — mirrors how pdg-query.test.ts pins `fnLineOf` with a `C:/...` id. + describe('fnFileOf recovers a Windows-style (drive-colon) path', () => { + it('projects a downstream block whose id carries a C:/ drive path to its owning symbol', async () => { + const result = await backend.callTool('impact', { + target: 'winFn', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + const items = Object.values(result.byDepth as Record).flat(); + // The downstream Windows block `BasicBlock:C:/src/win.ts:21:0:0` owns + // `winUse` (0-based startLine 20). If `fnFileOf` split from the LEFT it + // would yield `C` (the drive letter) as the path and fail to resolve the + // owning symbol — surfacing it as unresolved instead. Correct split-from- + // right recovers `C:/src/win.ts` and resolves `winUse`. + const winUse = items.find((i: any) => i.name === 'winUse'); + expect(winUse).toBeDefined(); + expect(winUse.id).toBe('func:winUse'); + expect(winUse.filePath).toBe('C:/src/win.ts'); + }); + }); + + // ── No-body symbol parity (KTD6 × KTD8) ─────────────────────────────────── + describe('no-body symbol still yields a parity-shaped (non-LOW) result', () => { + it('an interface (no CFG body) returns the no-body note + well-formed empty shape', async () => { + const result = await backend.callTool('impact', { + target: 'IShape', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.error).toBeUndefined(); + expect(result.risk).not.toBe('LOW'); // never a confident "safe to refactor" + expect(result.note).toMatch(/no.*(body|block|dependence)/i); + // Parity fields are present & well-formed (empty), not undefined. + expect(result.byDepth).toEqual({}); + expect(result.byDepthCounts['1']).toBe(0); + expect(result.affected_processes).toEqual([]); + expect(result.affected_modules).toEqual([]); + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const fn = ( + id: string, + name: string, + startLine: number, + endLine: number, + type: 'Function' | 'Interface' | 'Constructor' = 'Function', + ) => { + if (type === 'Constructor') { + return adapter.executePrepared( + `CREATE (n:Constructor {id: $id, name: $name, filePath: $filePath, startLine: $startLine, endLine: $endLine, content: 'x', description: 'shape fixture'})`, + { id, name, filePath: F, startLine, endLine }, + ); + } + return adapter.executePrepared( + `CREATE (n:${type} {id: $id, name: $name, filePath: $filePath, startLine: $startLine, endLine: $endLine, isExported: true, content: 'x', description: 'shape fixture'})`, + { id, name, filePath: F, startLine, endLine }, + ); + }; + const block = (id: string, startLine: number, text: string) => + adapter.executePrepared( + `CREATE (b:BasicBlock {id: $id, filePath: $filePath, startLine: $startLine, endLine: $startLine, text: $text})`, + { id, filePath: F, startLine, text }, + ); + const edge = (type: 'CDG' | 'REACHING_DEF', src: string, dst: string, reason: string) => + adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: '${type}', confidence: 1.0, reason: $reason, step: 0}]->(b)`, + { src, dst, reason }, + ); + + // Functions (0-based symbol lines). + await fn('func:target', 'target', 10, 10); // window [11,11] ⇒ seed {S} + await fn('func:up', 'up', 4, 4); // P@6 + await fn('func:down', 'down', 19, 21); // D1@21,D2@22 + await fn('func:ctl', 'ctl', 29, 31); // K1@31,K2@32 + // Same-line collision: dupA and dupB BOTH start at 0-based line 40. + await fn('func:dupA', 'dupTarget', 40, 42); + await fn('func:dupB', 'dupTarget', 40, 42); + // No-body interface (no blocks). + await fn('func:IShape', 'IShape', 50, 52, 'Interface'); + // Constructor owner projection. + await fn('ctor:FlowThing', 'FlowThing.constructor', 55, 55, 'Constructor'); + + // Blocks. + await block(S, 11, 'const x = compute();'); + await block(P, 6, 'const seed = input();'); + await block(D1, 21, 'use(x);'); + await block(D2, 22, 'log(x);'); + await block(K1, 31, 'doA();'); + await block(K2, 32, 'doB();'); + await block(T, 41, 'dispatch();'); // owned by BOTH dupA & dupB + await block(CT, 56, 'this.value = x;'); // owned by Constructor + await block(U, 99, 'top-level-side-effect();'); // owned by NO symbol + + // RD chain (def→use): P → S → D1 → D2 + await edge('REACHING_DEF', P, S, 'seed'); + await edge('REACHING_DEF', S, D1, 'x'); + await edge('REACHING_DEF', D1, D2, 'x'); + await edge('REACHING_DEF', S, CT, 'ctor'); + // CDG chain: P(controller) → S → K1 → K2 → T(@dup line) → U(top-level) + await edge('CDG', P, S, 'T'); + await edge('CDG', S, K1, 'T'); + await edge('CDG', K1, K2, 'T'); + await edge('CDG', K2, T, 'T'); + await edge('CDG', T, U, 'T'); + + // ── Statement-anchored fixture `accum` (mode:'pdg' + line) ──────────────── + // Self-contained multi-statement fn at 0-based [70,80] (window [71,81]); a + // line range that does NOT overlap S@11 / D@21,22 / K@31,32 above, so its + // whole-symbol seed cannot pick up an unrelated block. Every dependence + // stays inside it, so a STATEMENT seed slices the dependent statements. + // 71: let sum = 0; (A) 72: for (…) { (B, criterion) + // 74: sum = sum + x; (C) 76: return sum; (D) + // CDG B→C; RD A→C, C→D ⇒ downstream from line 72 = {C@74, D@76}. + await fn('func:accum', 'accum', 70, 80); + const AccA = `BasicBlock:${F}:71:0:0`; // line 71 + const AccB = `BasicBlock:${F}:71:0:1`; // line 72 + const AccC = `BasicBlock:${F}:71:0:2`; // line 74 + const AccD = `BasicBlock:${F}:71:0:3`; // line 76 + await block(AccA, 71, 'let sum = 0;'); + await block(AccB, 72, 'for (const x of xs) {'); + await block(AccC, 74, 'sum = sum + x;'); + await block(AccD, 76, 'return sum;'); + await edge('CDG', AccB, AccC, 'loop'); + await edge('REACHING_DEF', AccA, AccC, 'sum'); + await edge('REACHING_DEF', AccC, AccD, 'sum'); + + // ── Windows drive-colon path fixture (exercises fnFileOf split-from-right) ─ + // Separate file `C:/src/win.ts`, isolated from `target`'s graph so the + // existing impactedCount/byDepth assertions are untouched. `winFn` is its + // own seed target; a downstream RD block (`:21:`) owns `winUse`. + const WF = 'C:/src/win.ts'; + const winFnSeed = `BasicBlock:${WF}:11:0:0`; // winFn@0-based[10,12] ⇒ window [11,13] + const winUseBlk = `BasicBlock:${WF}:21:0:0`; // winUse@0-based[20,20] ⇒ fnLine 21 + const winNode = ( + id: string, + name: string, + startLine: number, + endLine: number, + type: 'Function' = 'Function', + ) => + adapter.executePrepared( + `CREATE (n:${type} {id: $id, name: $name, filePath: $filePath, startLine: $startLine, endLine: $endLine, isExported: true, content: 'x', description: 'win fixture'})`, + { id, name, filePath: WF, startLine, endLine }, + ); + const winBlock = (id: string, startLine: number, text: string) => + adapter.executePrepared( + `CREATE (b:BasicBlock {id: $id, filePath: $filePath, startLine: $startLine, endLine: $startLine, text: $text})`, + { id, filePath: WF, startLine, text }, + ); + await winNode('func:winFn', 'winFn', 10, 12); + await winNode('func:winUse', 'winUse', 20, 20); + await winBlock(winFnSeed, 11, 'const w = win();'); + await winBlock(winUseBlk, 21, 'useWin(w);'); + await edge('REACHING_DEF', winFnSeed, winUseBlk, 'w'); + + // ── Same-name-in-different-files fixture (FIX 1 keystone: seed must anchor + // on the file_path/uid-disambiguated symbol, NOT re-resolve by bare name) ── + // Two functions BOTH named `sameName`, in `src/a.ts` and `src/b.ts`, each + // with a DISTINCT downstream owner (`onlyA` vs `onlyB`). A correct seed + // (anchored on the already-resolved symbol's file+span) reaches only the + // chosen file's downstream block; a re-resolution by bare `sameName` would + // either re-ambiguate or anchor on the wrong file. + const AFILE = 'src/a.ts'; + const BFILE = 'src/b.ts'; + const sameSeedA = `BasicBlock:${AFILE}:11:0:0`; // sameName@A 0-based[10,10] ⇒ window [11,11] + const onlyABlk = `BasicBlock:${AFILE}:21:0:0`; // onlyA@0-based[20,20] ⇒ fnLine 21 + const sameSeedB = `BasicBlock:${BFILE}:11:0:0`; // sameName@B 0-based[10,10] ⇒ window [11,11] + const onlyBBlk = `BasicBlock:${BFILE}:21:0:0`; // onlyB@0-based[20,20] ⇒ fnLine 21 + const node2 = ( + id: string, + name: string, + filePath: string, + startLine: number, + endLine: number, + ) => + adapter.executePrepared( + `CREATE (n:Function {id: $id, name: $name, filePath: $filePath, startLine: $startLine, endLine: $endLine, isExported: true, content: 'x', description: 'samename fixture'})`, + { id, name, filePath, startLine, endLine }, + ); + const block2 = (id: string, filePath: string, startLine: number, text: string) => + adapter.executePrepared( + `CREATE (b:BasicBlock {id: $id, filePath: $filePath, startLine: $startLine, endLine: $startLine, text: $text})`, + { id, filePath, startLine, text }, + ); + await node2('func:sameA', 'sameName', AFILE, 10, 10); + await node2('func:onlyA', 'onlyA', AFILE, 20, 20); + await node2('func:sameB', 'sameName', BFILE, 10, 10); + await node2('func:onlyB', 'onlyB', BFILE, 20, 20); + await block2(sameSeedA, AFILE, 11, 'const a = mk();'); + await block2(onlyABlk, AFILE, 21, 'useA(a);'); + await block2(sameSeedB, BFILE, 11, 'const b = mk();'); + await block2(onlyBBlk, BFILE, 21, 'useB(b);'); + await edge('REACHING_DEF', sameSeedA, onlyABlk, 'a'); + await edge('REACHING_DEF', sameSeedB, onlyBBlk, 'b'); + + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'shape-repo', + path: '/shape/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'shape123', + stats: { files: 1, nodes: 18, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/impact-pdg-statement-precise.test.ts b/gitnexus/test/integration/impact-pdg-statement-precise.test.ts new file mode 100644 index 000000000..1cc348c25 --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-statement-precise.test.ts @@ -0,0 +1,121 @@ +/** + * Integration test: statement-precise inter-procedural reach (the proven subset). + * + * Gates the PR #2227 fix that a callee invoked DIRECTLY on the seeded line is + * proven (`statementPreciseByDepth`), not dropped — the seed block is excluded + * from `reachableBlocks` by the seed-minus-reachable convention, so the dispatch + * must union the seed block's callees. End-to-end against a real LadybugDB. + * + * Fixture shape (function `caller`, downstream seed on line 2): + * - B0 (line 2, seed): calls `foo` — `foo` is NOT called from any other block + * - B1 (line 4): calls `bar` — reachable from B0 via REACHING_DEF + * - B2 (line 5): calls `baz` — NOT reachable from B0 (no dependence edge) + * So the slice = {B0, B1}; proven = {foo, bar}; `baz` (reached in the call graph + * but not from the slice) stays unproven. + */ +import { it, expect, beforeAll, vi } from 'vitest'; +import type { RepoMeta } from '../../src/storage/repo-manager.js'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos, loadMeta } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + loadMeta: vi.fn().mockResolvedValue(null), + }; +}); + +const fn = (id: string, name: string, startLine: number, endLine: number): string => + `CREATE (:Function {id: '${id}', name: '${name}', filePath: 'src/fn.ts', startLine: ${startLine}, endLine: ${endLine}, isExported: true, content: 'x', description: 'statement-precise fixture'})`; + +const blk = (idx: number, line: number, callees: string): string => + `CREATE (:BasicBlock {id: 'BasicBlock:src/fn.ts:1:0:${idx}', filePath: 'src/fn.ts', startLine: ${line}, endLine: ${line}, text: 't', callees: '${callees}'})`; + +const calls = (callee: string): string => + `MATCH (c:Function {id: 'func:caller'}), (t:Function {id: '${callee}'}) CREATE (c)-[:CodeRelation {type: 'CALLS', confidence: 0.9, reason: 'local-call', step: 0}]->(t)`; + +const SEED: string[] = [ + fn('func:caller', 'caller', 0, 6), + fn('func:foo', 'foo', 10, 12), + fn('func:bar', 'bar', 14, 16), + fn('func:baz', 'baz', 18, 20), + blk(0, 2, 'foo'), // seed block (line 2) — calls foo + blk(1, 4, 'bar'), // dependent block (line 4) — calls bar + blk(2, 5, 'baz'), // unreachable block (line 5) — calls baz + // B0 -> B1: a downstream data dependence so B1 is reachable from the line-2 seed. + `MATCH (a:BasicBlock {id: 'BasicBlock:src/fn.ts:1:0:0'}), (b:BasicBlock {id: 'BasicBlock:src/fn.ts:1:0:1'}) CREATE (a)-[:CodeRelation {type: 'REACHING_DEF', confidence: 1.0, reason: 'v', step: 0}]->(b)`, + calls('func:foo'), + calls('func:bar'), + calls('func:baz'), +]; + +const READY_META: RepoMeta = { + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }, +} as unknown as RepoMeta; + +let backend: LocalBackend; + +withTestLbugDB( + 'impact-pdg-statement-precise', + () => { + beforeAll(() => { + if (!backend) throw new Error('LocalBackend not initialized in afterSetup'); + }); + + const provenNames = (result: { + pdgInterprocedural?: { statementPreciseByDepth?: Record> }; + }): string[] => + Object.values(result.pdgInterprocedural?.statementPreciseByDepth ?? {}) + .flat() + .map((item) => item.name ?? ''); + + it('proves the seed-line callee and the dependent callee; excludes the unreachable callee', async () => { + vi.mocked(loadMeta).mockResolvedValueOnce(READY_META); + const result = await backend.callTool('impact', { + target: 'caller', + direction: 'downstream', + mode: 'pdg', + line: 2, + }); + + expect(result.error).toBeUndefined(); + expect(result.epistemic).toBe('pdg-intra-procedural'); + const proven = provenNames(result); + // foo is called ON the seeded line (regression target); bar from the + // dependent block — both are statement-precise proven. + expect(proven).toContain('foo'); + expect(proven).toContain('bar'); + // baz is reached in the call graph but only from an unreachable block, so + // it is NOT in the statement-precise slice. + expect(proven).not.toContain('baz'); + // The full reach still lists all three (recall preserved). + const fullNames = Object.values(result.pdgInterprocedural?.byDepth ?? {}) + .flat() + .map((item: { name?: string }) => item.name ?? ''); + expect(fullNames).toEqual(expect.arrayContaining(['foo', 'bar', 'baz'])); + }); + }, + { + seed: SEED, + poolAdapter: true, + afterSetup: async (handle) => { + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'stmt-precise-repo', + path: '/stmt-precise/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'sp123', + stats: { files: 1, nodes: 7, communities: 0, processes: 0 }, + }, + ]); + backend = new LocalBackend(); + await backend.init(); + }, + }, +); diff --git a/gitnexus/test/integration/impact-pdg-traversal.test.ts b/gitnexus/test/integration/impact-pdg-traversal.test.ts new file mode 100644 index 000000000..cec8a6744 --- /dev/null +++ b/gitnexus/test/integration/impact-pdg-traversal.test.ts @@ -0,0 +1,486 @@ +/** + * Integration Tests: `impact` PDG-mode blast-radius TRAVERSAL (U3 / KTD2,4,6,11) + * + * End-to-end against a REAL LadybugDB, through the full `callTool('impact', …)` + * dispatch with `mode:'pdg'`. Exercises `_runImpactPDG` — the direction-aware + * bounded BFS over CDG + REACHING_DEF block edges — the correctness keystone of + * the feature (the KTD4 direction × edge-type truth table). + * + * The result exposes the consumer-safe impact shape plus traversal details + * (`reachableBlocks`, `truncated`, `depthReached`) so this suite can pin the + * graph algorithm without bypassing the public `impact` tool contract. + * + * ── Fixture graph (hand-seeded, no parser; controlled line numbers) ────────── + * One file `src/flow.ts`. The TARGET symbol `target` is a one-line function at + * 0-based symbol lines [10,10] ⇒ anchor window [11,11], fnLine segment '11'. Its + * single seed block `S` sits at 1-based line 11. All OTHER blocks belong to + * neighbouring functions OUTSIDE that window, so the seed set is exactly {S} and + * every reached block is unambiguously a traversal result, not a co-seed. + * + * RD (def→use, forward = downstream): + * P -[RD]-> S -[RD]-> D1 -[RD]-> D2 + * CDG (controller→dependent, forward = downstream): + * C -[CDG]-> S -[CDG]-> K1 -[CDG]-> K2 + * + * So from S: + * downstream RD → {D1, D2} (forward; NOT P) + * upstream RD → {P} (reverse; NOT D1/D2) + * downstream CDG → {K1, K2} (forward; NOT C) + * upstream CDG → {C} (reverse; NOT K1/K2) + * downstream (combined) → {D1,D2,K1,K2} (union, same forward sense) + * upstream (combined) → {P, C} (union, same reverse sense) + * + * `loadMeta` is mocked to stamp BOTH caps so `pdgLayerStatus` returns `ready` + * and the call falls through to the real traversal (U2's gate is exercised by + * impact-pdg-degradation.test.ts; here we drive past it). + */ +import { describe, it, expect, beforeAll, vi } from 'vitest'; +import type { RepoMeta } from '../../src/storage/repo-manager.js'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + // Both caps stamped ⇒ pdgLayerStatus === 'ready' ⇒ traversal runs. + loadMeta: vi.fn().mockResolvedValue({ + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }, + } as unknown as RepoMeta), + }; +}); + +const F = 'src/flow.ts'; +// Block ids: BasicBlock:::: +const S = `BasicBlock:${F}:11:0:0`; // target `target`'s only block (line 11) +const P = `BasicBlock:${F}:5:0:0`; // predecessor (fn `up`, line 6) — RD def & CDG controller into S +const D1 = `BasicBlock:${F}:20:0:0`; // RD use of S (fn `down`, line 21) +const D2 = `BasicBlock:${F}:20:0:1`; // RD use of D1 (line 22) +const K1 = `BasicBlock:${F}:30:0:0`; // CDG dependent of S (fn `ctl`, line 31) +const K2 = `BasicBlock:${F}:30:0:1`; // CDG dependent of K1 (line 32) + +withTestLbugDB( + 'impact-pdg-traversal', + (handle) => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized in afterSetup'); + backend = ext._backend; + }); + + const reachable = (result: any): string[] => + [...((result?.reachableBlocks as string[]) ?? [])].sort(); + + describe('KTD4 direction × edge-type truth table', () => { + it('downstream REACHING_DEF reaches the use blocks, not the def predecessor', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + const set = reachable(result); + // forward over RD: S→D1→D2 are reached; P (S's def predecessor) is NOT. + expect(set).toContain(D1); + expect(set).toContain(D2); + expect(set).not.toContain(P); + }); + + it('upstream REACHING_DEF reaches the defs reaching the target, not its uses', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'upstream', + mode: 'pdg', + }); + expect(result.error).toBeUndefined(); + const set = reachable(result); + // reverse over RD: P (def reaching S) reached; D1/D2 (uses) NOT. + expect(set).toContain(P); + expect(set).not.toContain(D1); + expect(set).not.toContain(D2); + }); + + it('downstream CDG reaches the controlled blocks, never the controller', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'downstream', + mode: 'pdg', + }); + const set = reachable(result); + // forward over CDG: S controls K1 controls K2. + expect(set).toContain(K1); + expect(set).toContain(K2); + // CDG-precision exclusion: the CONTROLLER `P` (the CDG predecessor of S) + // must NOT appear downstream — it is an *upstream* block. A CDG + // direction/precision bug (e.g. traversing the CDG edge in reverse) would + // leak P into the downstream set; pin it so such a bug fails here. + // (D1/D2 are legitimately present downstream via the REACHING_DEF edges — + // the combined CDG+RD frontier is one direction over BOTH edge types — so + // the meaningful CDG-precision exclusion is the controller, not the RD + // uses; the combined-frontier test below pins the exact {D1,D2,K1,K2} set.) + expect(set).not.toContain(P); + }); + + it('upstream CDG reaches the controller, not the controlled blocks', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'upstream', + mode: 'pdg', + }); + const set = reachable(result); + // reverse over CDG: C (controller of S) reached; K1/K2 (controlled) NOT. + expect(set).toContain(P); + expect(set).not.toContain(K1); + expect(set).not.toContain(K2); + }); + + it('combined CDG+RD downstream frontier is the forward union of both', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'downstream', + mode: 'pdg', + }); + const set = reachable(result); + expect(set).toEqual([D1, D2, K1, K2].sort()); + }); + + it('combined CDG+RD upstream frontier is the reverse union of both', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'upstream', + mode: 'pdg', + }); + const set = reachable(result); + // P is both the RD def AND the CDG controller of S, so the reverse union + // is exactly {P}; the forward successors D1/D2/K1/K2 are never reached. + expect(set).toEqual([P]); + }); + }); + + // Helper: the slice statements' lines (sorted) for an `accum` line-seeded call. + const sliceLines = (result: any): number[] => + [...((result?.affectedStatements as any[]) ?? [])].map((s) => s.line).sort((a, b) => a - b); + + describe('statement-anchored seed (mode:pdg + line)', () => { + it('downstream from line 72 returns exactly the statements dependent on it (NOT the whole symbol)', async () => { + const result = await backend.callTool('impact', { + target: 'accum', + direction: 'downstream', + mode: 'pdg', + line: 72, + }); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(result.criterionLine).toBe(72); + // line 72 (the loop header B) controls C@74, whose sum flows to D@76. + expect(sliceLines(result)).toEqual([74, 76]); + expect(result.affectedStatementCount).toBe(2); + // The dependent statements carry the real source line + text. + const byLine = new Map((result.affectedStatements as any[]).map((s) => [s.line, s])); + expect(byLine.get(74).text).toBe('sum = sum + x;'); + expect(byLine.get(74).filePath).toBe(F); + expect(byLine.get(76).text).toBe('return sum;'); + // It is NOT the whole-symbol set — line 71 (the def above the criterion) is + // upstream of the criterion, not downstream of it, so it must be absent. + expect(sliceLines(result)).not.toContain(71); + }); + + it('upstream from line 74 returns the statements line 74 depends on (the def + the controller)', async () => { + const result = await backend.callTool('impact', { + target: 'accum', + direction: 'upstream', + mode: 'pdg', + line: 74, + }); + expect(result.error).toBeUndefined(); + expect(result.criterionLine).toBe(74); + // C@74 depends on A@71 (RD def of sum) and B@72 (CDG controller). + expect(sliceLines(result)).toEqual([71, 72]); + expect(result.affectedStatementCount).toBe(2); + // NOT the whole-symbol set — D@76 is downstream of line 74, never upstream. + expect(sliceLines(result)).not.toContain(76); + }); + + it('whole-symbol (no line) is empty and steers the caller to line:', async () => { + // `accum`'s entire dependence stays inside its own [7,13] window, so a + // whole-symbol seed reaches nothing (every block is a co-seed) — the + // structurally-empty WHOLE-SYMBOL case. + const result = await backend.callTool('impact', { + target: 'accum', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + // No criterionLine (whole-symbol mode), an empty slice, and the steering note. + expect(result.criterionLine).toBeUndefined(); + expect(result.affectedStatements).toEqual([]); + expect(result.affectedStatementCount).toBe(0); + expect(result.note).toMatch(/WHOLE-SYMBOL/); + expect(result.note).toMatch(/line:|Pass line/i); + // Still never a confident "safe" zero. + expect(result.risk).not.toBe('LOW'); + }); + + it('a line with no statement block → epistemic pdg-no-block-at-line (distinct from no-pdg-body)', async () => { + const result = await backend.callTool('impact', { + target: 'accum', + direction: 'downstream', + mode: 'pdg', + line: 73, // blank line inside accum — no block starts here + }); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(result.criterionLine).toBe(73); + // Distinct from the no-PDG-body epistemic (the line has no statement block). + expect(result.epistemic).toBe('pdg-no-block-at-line'); + expect(result.epistemic).not.toBe('no-pdg-body'); + expect(result.affectedStatements).toEqual([]); + expect(result.risk).not.toBe('LOW'); + }); + }); + + describe('truncation signalling', () => { + it('maxDepth=1 truncates the chain and flags truncated (not silently short)', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'downstream', + mode: 'pdg', + maxDepth: 1, + }); + const set = reachable(result); + // Only the first hop: D1 and K1, NOT D2/K2 (depth 2). + expect(set).toContain(D1); + expect(set).toContain(K1); + expect(set).not.toContain(D2); + expect(set).not.toContain(K2); + expect(result.truncated).toBe(true); + expect(result.depthReached).toBe(1); + }); + + it('full traversal completing within depth is NOT flagged truncated', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'downstream', + mode: 'pdg', + maxDepth: 10, + }); + expect(result.truncated).toBeFalsy(); + }); + + it('an exact limit-sized seed/step is not flagged truncated without an extra row', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'upstream', + mode: 'pdg', + maxDepth: 10, + limit: 1, + }); + expect(reachable(result)).toEqual([P]); + expect(result.truncated).toBeFalsy(); + expect(result.truncatedBy).toBeUndefined(); + expect(result.truncatedByReasons).toBeUndefined(); + }); + + it('limit truncation bounds the reachable set and flags truncated', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'downstream', + mode: 'pdg', + maxDepth: 10, + limit: 1, + }); + // A limit of 1 cannot expand the full union — the result is bounded and + // flagged so a caller never reads the clipped set as the whole radius. + expect(reachable(result).length).toBeLessThan(4); + expect(result.truncated).toBe(true); + }); + + it('reports both depth and limit when both bounds truncate the slice', async () => { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'downstream', + mode: 'pdg', + maxDepth: 1, + limit: 1, + }); + expect(result.truncated).toBe(true); + expect(result.truncatedBy).toBe('depth'); + expect(result.truncatedByReasons).toEqual(['depth', 'limit']); + }); + + it('rejects or clamps a negative / huge / NaN limit (validated int interpolation)', async () => { + for (const limit of [-1, NaN, 1.5]) { + const result = await backend.callTool('impact', { + target: 'target', + direction: 'downstream', + mode: 'pdg', + limit, + }); + // Either a clean validation error OR a clamp to a sane default — but + // NEVER an unbounded/garbage interpolation or a crash. + if (result.error) { + expect(result.error).toMatch(/limit/i); + } else { + // Clamped: the traversal still produced its normal union. + expect(reachable(result).length).toBeGreaterThan(0); + } + } + }); + }); + + describe('KTD6 no-body symbol contract', () => { + it('a symbol with no BasicBlocks returns an explicit note, never a confident zero', async () => { + const result = await backend.callTool('impact', { + target: 'IShape', // interface — resolves, but has no CFG body / blocks + direction: 'downstream', + mode: 'pdg', + }); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + // Explicit "no PDG body for this symbol kind" signal … + expect(result.note).toMatch(/no.*(body|block|dependence)/i); + // … and NEVER a silent confident-LOW / impactedCount:0 with no marker. + expect(result.risk).not.toBe('LOW'); + expect(reachable(result)).toEqual([]); + }); + }); + + describe('KTD11 injection safety', () => { + it('a target containing a quote/colon is bound, not interpolated (no crash, no injection)', async () => { + // A malicious-looking target must flow through a bind param. It simply + // resolves to not-found here (no such symbol) — never a Cypher error. + const result = await backend.callTool('impact', { + target: `evil':' OR 1=1 //`, + direction: 'downstream', + mode: 'pdg', + }); + // Not a syntax/crash error — a clean not-found (param-bound). + expect(result.error).toMatch(/not found/i); + }); + }); + + describe('KTD2 ambiguous-anchor names the impact tool', () => { + it("an ambiguous target under mode:'pdg' names `impact`, not `pdg_query`", async () => { + const result = await backend.callTool('impact', { + target: 'dupTarget', // two functions share this name + direction: 'downstream', + mode: 'pdg', + }); + expect(result.status).toBe('ambiguous'); + // The ambiguous path is the _impactImpl one (U1), which already names the + // tool correctly; the load-bearing U3 fact is that resolveBlockAnchor's + // widened union never emits a `pdg_query` message for an impact call. + const blob = JSON.stringify(result); + expect(blob).not.toMatch(/pdg_query/); + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const fn = ( + id: string, + name: string, + startLine: number, + endLine: number, + type: 'Function' | 'Interface' = 'Function', + ) => + adapter.executePrepared( + `CREATE (n:${type} {id: $id, name: $name, filePath: $filePath, startLine: $startLine, endLine: $endLine, isExported: true, content: 'x', description: 'traversal fixture'})`, + { id, name, filePath: F, startLine, endLine }, + ); + const block = (id: string, startLine: number, text: string) => + adapter.executePrepared( + `CREATE (b:BasicBlock {id: $id, filePath: $filePath, startLine: $startLine, endLine: $startLine, text: $text})`, + { id, filePath: F, startLine, text }, + ); + const edge = (type: 'CDG' | 'REACHING_DEF', src: string, dst: string, reason: string) => + adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: '${type}', confidence: 1.0, reason: $reason, step: 0}]->(b)`, + { src, dst, reason }, + ); + + // Functions (0-based symbol lines). `target` is the one-line seed fn. + await fn('func:target', 'target', 10, 10); // window [11,11] ⇒ seed {S} + await fn('func:up', 'up', 4, 4); // P at line 6 — outside [11,11] + await fn('func:down', 'down', 19, 21); // D1/D2 — outside [11,11] + await fn('func:ctl', 'ctl', 29, 31); // K1/K2 — outside [11,11] + // No-body symbol: an interface with NO BasicBlocks at all. + await fn('func:IShape', 'IShape', 40, 42, 'Interface'); + // Ambiguous: two functions sharing a name. + await fn('func:dupTarget@a', 'dupTarget', 50, 52); + await fn('func:dupTarget@b', 'dupTarget', 60, 62); + + // Blocks. + await block(S, 11, 'const x = compute();'); // target's seed block + await block(P, 6, 'const seed = input();'); // predecessor + await block(D1, 21, 'use(x);'); // RD use + await block(D2, 22, 'log(x);'); // RD use of D1 + await block(K1, 31, 'doA();'); // CDG dependent + await block(K2, 32, 'doB();'); // CDG dependent of K1 + + // RD chain (def→use): P → S → D1 → D2 + await edge('REACHING_DEF', P, S, 'seed'); + await edge('REACHING_DEF', S, D1, 'x'); + await edge('REACHING_DEF', D1, D2, 'x'); + // CDG chain (controller→dependent): C(=P) → S → K1 → K2 + await edge('CDG', P, S, 'T'); + await edge('CDG', S, K1, 'T'); + await edge('CDG', K1, K2, 'T'); + + // ── Statement-anchored fixture `accum` (mode:'pdg' + line) ──────────────── + // A SELF-CONTAINED multi-statement function whose every dependence stays + // inside its own [71,81] window — so a WHOLE-SYMBOL seed reaches nothing + // (every block is a co-seed) while a STATEMENT seed (line N) yields exactly + // the statements dependent on line N. Lives in a line range that does NOT + // overlap the `target`/`up`/`down`/`ctl` blocks above, so its whole-symbol + // seed cannot pick up an unrelated block (the window is `[startLine+1, + // endLine+1]`). Mirrors the accumulator idiom: + // 71: let sum = 0; (A — RD def of sum) + // 72: for (const x of xs) { (B — CDG controller of the loop body) + // 73: (blank — NO block, the no-block-at-line case) + // 74: sum = sum + x; (C — accumulate; controlled by B, uses A) + // 76: return sum; (D — RD use of C's sum def) + // Edges (all intra-`accum`): + // CDG: B(72) → C(74) the loop controls the accumulate body + // RD: A(71) → C(74) sum's initial def reaches the accumulate use + // RD: C(74) → D(76) the accumulated sum flows to the return + // ⇒ downstream from line 72 = {C@74, D@76}; upstream from line 74 = {A@71, B@72}. + await fn('func:accum', 'accum', 70, 80); // window [71,81] + const AccA = `BasicBlock:${F}:71:0:0`; // line 71 + const AccB = `BasicBlock:${F}:71:0:1`; // line 72 (same fn, distinct blockIdx) + const AccC = `BasicBlock:${F}:71:0:2`; // line 74 + const AccD = `BasicBlock:${F}:71:0:3`; // line 76 + await block(AccA, 71, 'let sum = 0;'); + await block(AccB, 72, 'for (const x of xs) {'); + await block(AccC, 74, 'sum = sum + x;'); + await block(AccD, 76, 'return sum;'); + await edge('CDG', AccB, AccC, 'loop'); + await edge('REACHING_DEF', AccA, AccC, 'sum'); + await edge('REACHING_DEF', AccC, AccD, 'sum'); + + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'traversal-repo', + path: '/traversal/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'trav123', + stats: { files: 1, nodes: 12, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/route-method-roundtrip.test.ts b/gitnexus/test/integration/route-method-roundtrip.test.ts new file mode 100644 index 000000000..155bbb86b --- /dev/null +++ b/gitnexus/test/integration/route-method-roundtrip.test.ts @@ -0,0 +1,81 @@ +/** + * Real-LadybugDB round trip for `Route.method` (issue #2138, Part 1). + * + * This is the test the mocked `http-route-graph-method.test.ts` could NOT + * provide: it persists a `Route` node carrying `method` through the actual + * CSV generator + `COPY` path into a real LadybugDB, then runs the exact + * production `HANDLES_ROUTE_QUERY` and asserts the verb comes back. + * + * Before the schema/CSV/COPY columns were added, `HANDLES_ROUTE_QUERY`'s + * `route.method AS routeMethod` failed to bind against the real schema + * (`Binder exception: Cannot find property method for r.`) and the + * extractor's `catch { return [] }` silently swallowed it — so this test + * would have failed (empty rows / throw), pinning the exact regression. + * + * Coverage spans all three persistence points touched by Part 1: + * - `ROUTE_SCHEMA` (schema.ts) — the `method` column must exist + * - the Route CSV row (csv-generator.ts) — the value must be written + * - `getCopyQuery('Route')` (lbug-adapter.ts) — the COPY must load it + */ +import { it, expect } from 'vitest'; +import path from 'path'; +import fs from 'fs/promises'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; +import { buildTestGraph } from '../helpers/test-graph.js'; +import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.js'; +import { HANDLES_ROUTE_QUERY } from '../../src/core/group/extractors/http-route-extractor.js'; + +withTestLbugDB('route-method-roundtrip', (handle) => { + it('persists Route.method through CSV→COPY and HANDLES_ROUTE_QUERY returns it', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + // 1. Build a graph with a single framework Route node carrying `method`, + // mirroring what the routes phase now emits for a Spring controller. + const graph = buildTestGraph([ + { + id: 'Route:/api/orders', + label: 'Route', + name: '/api/orders', + filePath: 'OrderController.java', + extra: { + method: 'POST', + responseKeys: [], + errorKeys: [], + middleware: [], + }, + }, + ]); + + // 2. Generate CSVs through the real generator (exercises the new + // `method` column in the Route CSV row). + const csvDir = path.join(handle.tmpHandle.dbPath, 'csv-roundtrip'); + const repoDir = path.join(handle.tmpHandle.dbPath, 'repo-roundtrip'); + await fs.mkdir(repoDir, { recursive: true }); + await streamAllCSVsToDisk(graph, repoDir, csvDir); + + // Sanity: the generated route.csv header + row include the method column. + const routeCsv = await fs.readFile(path.join(csvDir, 'route.csv'), 'utf-8'); + expect(routeCsv.split('\n')[0]).toContain('method'); + expect(routeCsv).toContain('POST'); + + // 3. COPY the Route node into the real DB via the production COPY query + // (exercises the new `method` column in getCopyQuery('Route')). + const routeCsvPath = path.join(csvDir, 'route.csv').replace(/\\/g, '/'); + await adapter.executeQuery(adapter.getCopyQuery('Route', routeCsvPath)); + + // 4. Seed the handler File node + HANDLES_ROUTE edge via Cypher. + await adapter.executeQuery( + `CREATE (:File {id: 'File:OrderController.java', name: 'OrderController.java', filePath: 'OrderController.java'})`, + ); + await adapter.executeQuery( + `MATCH (f:File {id: 'File:OrderController.java'}), (r:Route {id: 'Route:/api/orders'}) + CREATE (f)-[:CodeRelation {type: 'HANDLES_ROUTE', confidence: 1.0, reason: 'framework-route', step: 0}]->(r)`, + ); + + // 5. Run the EXACT production query and assert the verb round-trips. + const rows = (await adapter.executeQuery(HANDLES_ROUTE_QUERY)) as Record[]; + const row = rows.find((r) => String(r.routePath) === '/api/orders'); + expect(row, 'HANDLES_ROUTE_QUERY returned no row for the seeded route').toBeTruthy(); + expect(row!.routeMethod).toBe('POST'); + }); +}); diff --git a/gitnexus/test/unit/ai-context.test.ts b/gitnexus/test/unit/ai-context.test.ts index 30a9441c1..c3eca3458 100644 --- a/gitnexus/test/unit/ai-context.test.ts +++ b/gitnexus/test/unit/ai-context.test.ts @@ -155,6 +155,9 @@ describe('generateAIContextFiles', () => { const withPdg = generateGitNexusContent('PdgProject', stats, { hasPdg: true }); expect(withPdg).toContain('pdg_query'); expect(withPdg).toContain('under what condition does X run'); + expect(withPdg).toContain('line: '); + expect(withPdg).toContain('affectedStatements'); + expect(withPdg).toContain('byDepth'); // hasPdg omitted (default false) → no pdg_query line; a non-pdg index must // not advertise a tool that only returns a "no PDG layer" note. const withoutPdg = generateGitNexusContent('PlainProject', stats); diff --git a/gitnexus/test/unit/basicblock-callee-ids-schema.test.ts b/gitnexus/test/unit/basicblock-callee-ids-schema.test.ts new file mode 100644 index 000000000..56c79a3b5 --- /dev/null +++ b/gitnexus/test/unit/basicblock-callee-ids-schema.test.ts @@ -0,0 +1,307 @@ +/** + * U4 (#2227 follow-up plan) — `BasicBlock.calleeIds` column wiring end-to-end. + * + * U3 (already committed) sets `node.properties.calleeIds` on every emitted + * BasicBlock; U4 wires that property through the FIVE places the existing + * `callees` column lives so it actually persists and reads back: + * 1. schema DDL (BASICBLOCK_SCHEMA) + * 2. CSV header + row builder (BASICBLOCK_CSV_HEADER / buildBasicBlockRow) + * 3. bulk COPY column list (getCopyQuery('BasicBlock')) + * 4. single-node CREATE (insertNodeToLbug) + * 5. incremental MERGE (batchInsertNodesToLbug) + * plus the INCREMENTAL_SCHEMA_VERSION 2 → 3 bump (KTD5). + * + * `calleeIds` is added LAST in the CSV/COPY/CREATE/MERGE tuple, so the column + * order MUST stay identical across header, COPY list, and row array — the + * parity test below is the drift guard. + * + * The CSV/COPY/schema/version assertions are pure (no DB). The CREATE and MERGE + * query-string assertions mock `lbug-config.js` to capture the executed cypher, + * mirroring the established `lbug-adapter-wal-schema.test.ts` pattern (typed + * fake `conn`/`db`, no `any`/`as any`). + */ +import { afterEach, describe, expect, it, vi } from 'vitest'; +import type { GraphNode, NodeProperties } from 'gitnexus-shared'; +import { BASICBLOCK_CSV_HEADER, buildBasicBlockRow } from '../../src/core/lbug/csv-generator.js'; +import { getCopyQuery } from '../../src/core/lbug/lbug-adapter.js'; +import { BASICBLOCK_SCHEMA } from '../../src/core/lbug/schema.js'; +import { INCREMENTAL_SCHEMA_VERSION } from '../../src/storage/repo-manager.js'; + +// ── helpers ───────────────────────────────────────────────────────────────── + +/** Build a typed BasicBlock GraphNode with the given extra properties. */ +const basicBlock = (id: string, props: Partial): GraphNode => ({ + id, + label: 'BasicBlock', + properties: { + name: '', // BasicBlock has no name column; identified by id + span + filePath: 'src/a.ts', + startLine: 1, + endLine: 3, + text: 'foo(); bar();', + ...props, + }, +}); + +/** Parse the COPY column tuple out of `COPY Table(a, b, c) FROM "…"`. */ +const copyColumns = (copyQuery: string): string[] => { + const open = copyQuery.indexOf('('); + const close = copyQuery.indexOf(')', open); + return copyQuery + .slice(open + 1, close) + .split(',') + .map((c) => c.trim()); +}; + +// ── 1. CSV header / row builder (pure) ──────────────────────────────────────── + +describe('BasicBlock calleeIds — CSV header + row builder', () => { + it('header lists calleeIds LAST, after callees', () => { + expect(BASICBLOCK_CSV_HEADER).toBe('id,filePath,startLine,endLine,text,callees,calleeIds'); + const cols = BASICBLOCK_CSV_HEADER.split(','); + expect(cols[cols.length - 1]).toBe('calleeIds'); + expect(cols[cols.length - 2]).toBe('callees'); + }); + + it('round-trip: calleeIds lands in the column the header names for it', () => { + const node = basicBlock('BasicBlock:src/a.ts:0', { + callees: 'foo bar', + calleeIds: 'id1 id2', + }); + const headerCols = BASICBLOCK_CSV_HEADER.split(','); + const rowCells = buildBasicBlockRow(node).split(','); + + // Same arity as the header → positional column match is meaningful. + expect(rowCells).toHaveLength(headerCols.length); + const calleeIdsIdx = headerCols.indexOf('calleeIds'); + const calleesIdx = headerCols.indexOf('callees'); + // escapeCSVField always wraps the cell in double quotes; the space-joined + // id list contains no comma, so the cell is a single CSV column. + expect(rowCells[calleeIdsIdx]).toBe('"id1 id2"'); + expect(rowCells[calleesIdx]).toBe('"foo bar"'); + }); + + it('empty default: a node with no calleeIds property → empty cell, not "undefined"', () => { + const node = basicBlock('BasicBlock:src/a.ts:1', { callees: 'foo bar' }); + const headerCols = BASICBLOCK_CSV_HEADER.split(','); + const rowCells = buildBasicBlockRow(node).split(','); + const calleeIdsIdx = headerCols.indexOf('calleeIds'); + expect(rowCells[calleeIdsIdx]).toBe('""'); + expect(rowCells[calleeIdsIdx]).not.toContain('undefined'); + }); +}); + +// ── 2. Header / COPY tuple / row array PARITY (drift guard) ──────────────────── + +describe('BasicBlock calleeIds — header/COPY/row column parity', () => { + it('header column count == COPY tuple arity == row array length', () => { + const headerCols = BASICBLOCK_CSV_HEADER.split(','); + const copyCols = copyColumns(getCopyQuery('BasicBlock', '/tmp/bb.csv')); + const rowCells = buildBasicBlockRow( + basicBlock('BasicBlock:src/a.ts:0', { callees: 'foo', calleeIds: 'id1' }), + ).split(','); + + // Exact counts (the drift guard): 7 columns through and through. + expect(headerCols).toHaveLength(7); + expect(copyCols).toHaveLength(7); + expect(rowCells).toHaveLength(7); + expect(copyCols).toHaveLength(headerCols.length); + expect(rowCells).toHaveLength(headerCols.length); + }); + + it('COPY column order matches the CSV header order exactly', () => { + const headerCols = BASICBLOCK_CSV_HEADER.split(','); + const copyCols = copyColumns(getCopyQuery('BasicBlock', '/tmp/bb.csv')); + expect(copyCols).toEqual(headerCols); + expect(copyCols[copyCols.length - 1]).toBe('calleeIds'); + }); +}); + +// ── 3. schema DDL + incremental version bump (pure) ─────────────────────────── + +describe('BasicBlock calleeIds — schema DDL + version bump', () => { + it('BASICBLOCK_SCHEMA declares the calleeIds STRING column', () => { + expect(BASICBLOCK_SCHEMA).toContain('callees STRING'); + expect(BASICBLOCK_SCHEMA).toContain('calleeIds STRING'); + }); + + it('INCREMENTAL_SCHEMA_VERSION is at least 3 (calleeIds column bump, KTD5)', () => { + // The exact value advances as later milestones add re-index-forcing changes + // (v4 = CALL_SUMMARY, PDG FU-C). This guard pins the floor the calleeIds + // column established; the v3→4 reuse-gate guard lives in its own test. + expect(INCREMENTAL_SCHEMA_VERSION).toBeGreaterThanOrEqual(3); + }); +}); + +// ── 4 + 5. CREATE / MERGE query strings carry the calleeIds assignment ───────── +// +// `insertNodeToLbug` (CREATE) and `batchInsertNodesToLbug` (MERGE) build their +// cypher inline and execute it via the `lbug-config.js` connection. Mock that +// module with a typed fake `conn` whose `query` records the cypher, then assert +// the recorded string contains the calleeIds assignment. + +interface FakeQueryResult { + getAll: () => Promise; + close: () => void; +} +interface FakeConn { + query: (cypher: string) => Promise; + close: () => Promise; +} +interface FakeDb { + close: () => Promise; +} + +const makeConfigMock = () => { + const queries: string[] = []; + const queryResult: FakeQueryResult = { getAll: async () => [], close: vi.fn() }; + const conn: FakeConn = { + query: vi.fn(async (cypher: string) => { + queries.push(cypher); + return queryResult; + }), + close: vi.fn(async () => {}), + }; + const db: FakeDb = { close: vi.fn(async () => {}) }; + const mock = { + openLbugConnection: vi.fn(async () => ({ db, conn })), + closeLbugConnection: async (handle: { conn: FakeConn; db: FakeDb }) => { + await handle.conn.close(); + await handle.db.close(); + }, + isDbBusyError: vi.fn(() => false), + isOpenRetryExhausted: vi.fn(() => false), + isWalCorruptionError: vi.fn(() => false), + toNativeSafePath: (p: string) => p, + resolveNativeSafeStorageDir: (p: string) => p, + WAL_RECOVERY_SUGGESTION: 'run analyze --force', + waitForWindowsHandleRelease: vi.fn(async () => true), + }; + return { mock, queries }; +}; + +describe('BasicBlock calleeIds — CREATE / MERGE query strings', () => { + afterEach(() => { + vi.doUnmock('../../src/core/lbug/lbug-config.js'); + vi.resetModules(); + vi.clearAllMocks(); + }); + + it('insertNodeToLbug CREATE assigns calleeIds (after callees)', async () => { + vi.resetModules(); + const { mock, queries } = makeConfigMock(); + vi.doMock('../../src/core/lbug/lbug-config.js', () => mock); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const ok = await adapter.insertNodeToLbug( + 'BasicBlock', + { + id: 'BasicBlock:src/a.ts:0', + filePath: 'src/a.ts', + startLine: 1, + endLine: 3, + text: 'foo();', + callees: 'foo', + calleeIds: 'id1 id2', + }, + '/tmp/gitnexus-bb-create/lbug', + ); + + expect(ok).toBe(true); + const createQuery = queries.find((q) => q.startsWith('CREATE (n:BasicBlock')); + expect(createQuery).toBeDefined(); + expect(createQuery).toContain("calleeIds: 'id1 id2'"); + expect(createQuery).toContain("callees: 'foo'"); + // calleeIds is the LAST assignment in the tuple (mirrors the column order). + expect(createQuery?.indexOf('calleeIds:')).toBeGreaterThan( + createQuery?.indexOf('callees:') ?? -1, + ); + }); + + it('insertNodeToLbug CREATE defaults a missing calleeIds to an empty string', async () => { + vi.resetModules(); + const { mock, queries } = makeConfigMock(); + vi.doMock('../../src/core/lbug/lbug-config.js', () => mock); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.insertNodeToLbug( + 'BasicBlock', + { + id: 'BasicBlock:src/a.ts:0', + filePath: 'src/a.ts', + startLine: 1, + endLine: 3, + text: 'foo();', + callees: 'foo', + }, + '/tmp/gitnexus-bb-create-default/lbug', + ); + + const createQuery = queries.find((q) => q.startsWith('CREATE (n:BasicBlock')); + expect(createQuery).toContain("calleeIds: ''"); + expect(createQuery).not.toContain('calleeIds: undefined'); + }); + + it('batchInsertNodesToLbug MERGE sets n.calleeIds (after n.callees)', async () => { + vi.resetModules(); + const { mock, queries } = makeConfigMock(); + vi.doMock('../../src/core/lbug/lbug-config.js', () => mock); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const result = await adapter.batchInsertNodesToLbug( + [ + { + label: 'BasicBlock', + properties: { + id: 'BasicBlock:src/a.ts:0', + filePath: 'src/a.ts', + startLine: 1, + endLine: 3, + text: 'foo();', + callees: 'foo', + calleeIds: 'id1 id2', + }, + }, + ], + '/tmp/gitnexus-bb-merge/lbug', + ); + + expect(result.inserted).toBe(1); + const mergeQuery = queries.find((q) => q.startsWith('MERGE (n:BasicBlock')); + expect(mergeQuery).toBeDefined(); + expect(mergeQuery).toContain('n.calleeIds'); + expect(mergeQuery).toContain("n.calleeIds = 'id1 id2'"); + expect(mergeQuery).toContain("n.callees = 'foo'"); + expect(mergeQuery?.indexOf('n.calleeIds')).toBeGreaterThan( + mergeQuery?.indexOf('n.callees') ?? -1, + ); + }); + + it('batchInsertNodesToLbug MERGE defaults a missing calleeIds to an empty string', async () => { + vi.resetModules(); + const { mock, queries } = makeConfigMock(); + vi.doMock('../../src/core/lbug/lbug-config.js', () => mock); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.batchInsertNodesToLbug( + [ + { + label: 'BasicBlock', + properties: { + id: 'BasicBlock:src/a.ts:0', + filePath: 'src/a.ts', + startLine: 1, + endLine: 3, + text: 'foo();', + callees: 'foo', + }, + }, + ], + '/tmp/gitnexus-bb-merge-default/lbug', + ); + + const mergeQuery = queries.find((q) => q.startsWith('MERGE (n:BasicBlock')); + expect(mergeQuery).toContain("n.calleeIds = ''"); + expect(mergeQuery).not.toContain('n.calleeIds = undefined'); + }); +}); diff --git a/gitnexus/test/unit/blade-template-routes.test.ts b/gitnexus/test/unit/blade-template-routes.test.ts index 9edd6e791..305b1e01d 100644 --- a/gitnexus/test/unit/blade-template-routes.test.ts +++ b/gitnexus/test/unit/blade-template-routes.test.ts @@ -146,6 +146,7 @@ describe('Blade/template static route extraction', () => { expect(output.routeRegistry.get('/admin/orders')).toEqual({ filePath: 'routes/web.php', source: 'framework-route', + method: 'POST', }); const fetchEdges = graph.relationships.filter((rel) => rel.type === 'FETCHES'); diff --git a/gitnexus/test/unit/call-summary-schema-version.test.ts b/gitnexus/test/unit/call-summary-schema-version.test.ts new file mode 100644 index 000000000..dc092b87a --- /dev/null +++ b/gitnexus/test/unit/call-summary-schema-version.test.ts @@ -0,0 +1,93 @@ +/** + * PDG FU-C (U-C1 / U-C5) — CALL_SUMMARY relation-type posture + the v3→4 + * incremental reuse gate. + * + * CALL_SUMMARY is an INTERNAL PDG-engine edge: like the taint substrate edges + * (TAINTED / TAINT_PATH / CDG / REACHING_DEF / CFG) it must stay OUT of + * `VALID_RELATION_TYPES` so it never enters impact-style symbol-space traversal, + * and the impact relType allowlists (local-backend.ts ~:4373 / ~:5674) that gate + * on `VALID_RELATION_TYPES` therefore never surface it. The v4 bump forces a + * full re-analyze on a pre-v4 index (which has no CALL_SUMMARY edges, so an + * incremental top-up would silently under-report return-value ascent). + */ + +import { describe, it, expect } from 'vitest'; +import { readFileSync } from 'node:fs'; +import { fileURLToPath } from 'node:url'; +import path from 'node:path'; +import { + VALID_RELATION_TYPES, + EPISTEMIC_HERITAGE_RELATION_TYPES, + EPISTEMIC_CONSUMER_RELATION_TYPES, +} from '../../src/mcp/local/local-backend.js'; +import { INCREMENTAL_SCHEMA_VERSION } from '../../src/storage/repo-manager.js'; + +const here = path.dirname(fileURLToPath(import.meta.url)); +const repoRoot = path.resolve(here, '..', '..'); + +describe('CALL_SUMMARY relation-type exclusion (U-C1)', () => { + it('is NOT in VALID_RELATION_TYPES (never enters impact symbol-space traversal)', () => { + expect(VALID_RELATION_TYPES.has('CALL_SUMMARY')).toBe(false); + }); + + it('shares the internal-PDG-edge exclusion posture with the taint substrate edges', () => { + // The whole PDG/taint substrate stays out of the impact allowlist. + expect(VALID_RELATION_TYPES.has('TAINT_PATH')).toBe(false); + expect(VALID_RELATION_TYPES.has('TAINTED')).toBe(false); + expect(VALID_RELATION_TYPES.has('REACHING_DEF')).toBe(false); + expect(VALID_RELATION_TYPES.has('CFG')).toBe(false); + expect(VALID_RELATION_TYPES.has('CDG')).toBe(false); + // Sanity floor: the public callgraph edges ARE in the allowlist. + expect(VALID_RELATION_TYPES.has('CALLS')).toBe(true); + }); + + it('is absent from the epistemic-boundary relation sets', () => { + expect(EPISTEMIC_HERITAGE_RELATION_TYPES).not.toContain('CALL_SUMMARY'); + expect(EPISTEMIC_CONSUMER_RELATION_TYPES).not.toContain('CALL_SUMMARY'); + }); + + it('is absent from the impact relType default allowlists in local-backend (the ~:4373/~:5674 filters)', () => { + // The two impact relType filters first intersect with VALID_RELATION_TYPES + // (above) and otherwise fall back to a hardcoded public-edge default list. + // Assert CALL_SUMMARY appears in NEITHER default list's source text, so it + // can never be the relType an impact traversal walks. + const src = readFileSync( + path.join(repoRoot, 'src', 'mcp', 'local', 'local-backend.ts'), + 'utf8', + ); + // Every default relType array literal in the impact filters. + const defaultLists = src.match(/\[\s*\n\s*'CALLS',[\s\S]*?\]/g) ?? []; + expect(defaultLists.length).toBeGreaterThan(0); + for (const list of defaultLists) { + expect(list).not.toContain('CALL_SUMMARY'); + } + }); + + it('the /api/graph relationship projection does not special-case (allow OR block) CALL_SUMMARY', () => { + // The /api/graph relationship query (api.ts GRAPH_RELATIONSHIP_QUERY) is an + // unfiltered MATCH used for visualization, not an impact surface — it must + // not name CALL_SUMMARY in either direction (no bespoke allow/deny clause). + const api = readFileSync(path.join(repoRoot, 'src', 'server', 'api.ts'), 'utf8'); + expect(api).not.toContain('CALL_SUMMARY'); + }); +}); + +describe('CALL_SUMMARY incremental reuse gate (U-C5)', () => { + it('INCREMENTAL_SCHEMA_VERSION is bumped to 4 (CALL_SUMMARY re-index window)', () => { + expect(INCREMENTAL_SCHEMA_VERSION).toBe(4); + }); + + it('a pre-v4 (v3) stamp fails the `=== INCREMENTAL_SCHEMA_VERSION` reuse gate → forces full re-analyze', () => { + // The reuse gate at run-analyze.ts:920 is exactly this strict equality on + // the persisted `existingMeta.schemaVersion` (a plain number, possibly + // absent on a legacy stamp). Replicate it as a typed predicate. + const passesReuseGate = (stampedSchemaVersion: number | undefined): boolean => + stampedSchemaVersion === INCREMENTAL_SCHEMA_VERSION; + // A pre-v4 (v3) index has no CALL_SUMMARY edges → must NOT reuse → full re-analyze. + expect(passesReuseGate(3)).toBe(false); + // A legacy stamp with no schemaVersion at all is likewise rejected. + expect(passesReuseGate(undefined)).toBe(false); + // A current-version stamp passes the gate (incremental top-up eligible). + expect(passesReuseGate(4)).toBe(true); + }); +}); diff --git a/gitnexus/test/unit/calltool-dispatch-id-bridge.test.ts b/gitnexus/test/unit/calltool-dispatch-id-bridge.test.ts new file mode 100644 index 000000000..ca009c542 --- /dev/null +++ b/gitnexus/test/unit/calltool-dispatch-id-bridge.test.ts @@ -0,0 +1,298 @@ +/** + * Unit Tests: PDG impact dispatch — resolved-callee-id bridge (U6) + * + * Pins the U6 wiring that makes the U5 sound id-match actually fire end-to-end: + * 1. the dispatch loads the slice blocks' `BasicBlock.calleeIds` + * (`calleeIdsOfBlocks`) from the SAME seed ∪ reachable set as the names, and + * 2. the BFS evidence-stamping loop feeds the reached callee's resolved id + * (`relId`) into `pdgBridgeEvidenceForImpact`, so a first-hop callee is proven + * by `id ∈ sliceCalleeIds` (KTD3) rather than by leaf name. + * + * These drive the REAL `_runImpactBFS` (only `_runImpactPDG` is stubbed to supply + * the slice blocks) so the `calleeId` wiring in the stamping loop is exercised, and + * assert the verdict on the composed `interproceduralByDepth[depth][i].pdgEvidence`. + * + * Mocking mirrors test/unit/calltool-dispatch.test.ts exactly (same hoisted lbug + * mocks, same READY-PDG-layer `loadMeta` stamp, `executeParameterized` typed to its + * real signature via vi.mocked). + */ +import { describe, it, expect, vi, beforeEach } from 'vitest'; + +const { lbugMocks, platformMocks } = vi.hoisted(() => ({ + lbugMocks: { + initLbug: vi.fn().mockResolvedValue(undefined), + executeQuery: vi.fn().mockResolvedValue([]), + executeParameterized: vi.fn().mockResolvedValue([]), + closeLbug: vi.fn().mockResolvedValue(undefined), + isLbugReady: vi.fn().mockReturnValue(true), + }, + platformMocks: { + isVectorExtensionSupportedByPlatform: vi.fn().mockReturnValue(true), + }, +})); + +vi.mock('../../src/core/lbug/pool-adapter.js', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, ...lbugMocks }; +}); + +vi.mock('../../src/mcp/core/lbug-adapter.js', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, ...lbugMocks }; +}); + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + loadMeta: vi.fn(actual.loadMeta), + }; +}); + +vi.mock('../../src/core/git-staleness.js', () => ({ + checkStaleness: vi.fn().mockReturnValue({ isStale: false, commitsBehind: 0 }), + checkStalenessAsync: vi.fn().mockResolvedValue({ isStale: false, commitsBehind: 0 }), + checkCwdMatch: vi.fn().mockResolvedValue({ match: 'none' }), +})); + +vi.mock('../../src/storage/git.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + getGitRoot: vi.fn().mockReturnValue(null), + }; +}); + +vi.mock('../../src/core/platform/capabilities.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + isVectorExtensionSupportedByPlatform: platformMocks.isVectorExtensionSupportedByPlatform, + }; +}); + +vi.mock('../../src/core/search/bm25-index.js', () => ({ + searchFTSFromLbug: vi.fn().mockResolvedValue({ results: [], ftsAvailable: true }), +})); + +vi.mock('../../src/mcp/core/embedder.js', () => ({ + embedQuery: vi.fn().mockResolvedValue([]), + getEmbeddingDims: vi.fn().mockReturnValue(384), +})); + +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos, loadMeta } from '../../src/storage/repo-manager.js'; +import { executeParameterized } from '../../src/mcp/core/lbug-adapter.js'; + +// ─── Helpers ───────────────────────────────────────────────────────── + +const MOCK_REPO_ENTRY = { + name: 'test-project', + path: '/tmp/test-project', + storagePath: '/tmp/.gitnexus/test-project', + indexedAt: '2024-06-01T12:00:00Z', + lastCommit: 'abc1234567890', + stats: { files: 10, nodes: 50, edges: 100, communities: 3, processes: 5 }, +}; + +const TARGET_ROW = { + id: 'func:main', + name: 'main', + type: 'Function', + filePath: 'src/index.ts', +}; + +/** + * A `_runImpactPDG` stub result whose slice carries one reachable + one seed + * block — the dispatch queries both for `callees`/`calleeIds` (seed ∪ reachable). + */ +const PDG_SLICE_RESULT = { + mode: 'pdg', + target: TARGET_ROW, + direction: 'downstream', + risk: 'UNKNOWN', + impactedCount: 0, + epistemic: 'pdg-intra-procedural', + reachableBlocks: ['BasicBlock:src/index.ts:8:0:1'], + // Intra-only slice ⇒ the intra reach the bridge keys on equals reachableBlocks + // (FIX 6: bridge keys its first-hop-proven set on intraReachableBlocks). + intraReachableBlocks: ['BasicBlock:src/index.ts:8:0:1'], + seedBlocks: ['BasicBlock:src/index.ts:8:0:0'], + blockCount: 1, + affectedStatements: [{ line: 8, filePath: 'src/index.ts', text: 'callee()' }], + affectedStatementCount: 1, + criterionLine: 8, +}; + +/** Build a downstream BFS frontier row (one reached callee). */ +function frontierRow(id: string, name: string) { + return { + sourceId: 'func:main', + id, + name, + type: 'Function', + filePath: 'src/callee.ts', + relType: 'CALLS', + confidence: 0.9, + }; +} + +describe('LocalBackend PDG impact — resolved-callee-id bridge (U6)', () => { + let backend: LocalBackend; + + beforeEach(async () => { + vi.clearAllMocks(); + platformMocks.isVectorExtensionSupportedByPlatform.mockReturnValue(true); + // READY PDG layer so the dispatch reaches the mode-dispatch / bridge surface. + vi.mocked(loadMeta).mockResolvedValue({ + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }, + } as unknown as Awaited>); + vi.mocked(listRegisteredRepos).mockResolvedValue([MOCK_REPO_ENTRY] as unknown as Awaited< + ReturnType + >); + backend = new LocalBackend(); + await backend.init(); + }); + + it('proves a seed-line callee by resolved id (calleeIds slice match overrides name path)', async () => { + vi.spyOn( + backend as unknown as { _runImpactPDG: () => Promise }, + '_runImpactPDG', + ).mockResolvedValueOnce({ ...PDG_SLICE_RESULT }); + + // calleeIds carries the reached callee's id; callees carries a leaf name that + // does NOT match the reached callee's name → only the id path can prove it. + vi.mocked(executeParameterized).mockImplementation(async (_repo, query) => { + if (query.includes('RETURN b.calleeIds')) return [{ calleeIds: 'func:callee-A' }]; + if (query.includes('RETURN b.callees')) return [{ callees: 'someOtherLeaf' }]; + if (query.includes('r.type IN $relTypes') && !query.includes('STEP_IN_PROCESS')) { + return [frontierRow('func:callee-A', 'callee')]; + } + if (query.includes('COUNT(DISTINCT s.id)') || query.includes('RETURN s.id AS sid')) return []; + // Target resolution (WHERE n.name = $symName) and any other read. + return [TARGET_ROW]; + }); + + const result = await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + line: 8, + }); + + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + // The reached callee is proven by id ∈ sliceCalleeIds even though its leaf + // name is absent from sliceCalleeNames. + expect(result.interproceduralByDepth[1][0]).toMatchObject({ + id: 'func:callee-A', + pdgEvidence: 'callgraph-bridge', + }); + }); + + it('falls back to the name path when the calleeIds query errors (empty id set, no crash)', async () => { + vi.spyOn( + backend as unknown as { _runImpactPDG: () => Promise }, + '_runImpactPDG', + ).mockResolvedValueOnce({ ...PDG_SLICE_RESULT }); + + // The calleeIds query throws → calleeIdsOfBlocks swallows → empty id set → the + // bridge uses the name path (callees carries the reached callee's leaf name). + vi.mocked(executeParameterized).mockImplementation(async (_repo, query) => { + if (query.includes('RETURN b.calleeIds')) throw new Error('calleeIds query failed'); + if (query.includes('RETURN b.callees')) return [{ callees: 'callee' }]; + if (query.includes('r.type IN $relTypes') && !query.includes('STEP_IN_PROCESS')) { + return [frontierRow('func:callee-A', 'callee')]; + } + if (query.includes('COUNT(DISTINCT s.id)') || query.includes('RETURN s.id AS sid')) return []; + return [TARGET_ROW]; + }); + + const result = await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + line: 8, + }); + + // No crash, no surfaced error; the name path proves the reached callee. + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(result.interproceduralByDepth[1][0]).toMatchObject({ + id: 'func:callee-A', + pdgEvidence: 'callgraph-bridge', + }); + }); + + it('discriminates same-leaf-name callees by id end-to-end (only the in-slice id is proven)', async () => { + vi.spyOn( + backend as unknown as { _runImpactPDG: () => Promise }, + '_runImpactPDG', + ).mockResolvedValueOnce({ ...PDG_SLICE_RESULT }); + + // Two reached callees share the leaf name 'callee'. sliceCalleeIds carries ONLY + // func:callee-A → the id path proves A and leaves B unproven, where the name + // path (which sees 'callee' for both) would have over-attributed both. + vi.mocked(executeParameterized).mockImplementation(async (_repo, query) => { + if (query.includes('RETURN b.calleeIds')) return [{ calleeIds: 'func:callee-A' }]; + if (query.includes('RETURN b.callees')) return [{ callees: 'callee' }]; + if (query.includes('r.type IN $relTypes') && !query.includes('STEP_IN_PROCESS')) { + return [frontierRow('func:callee-A', 'callee'), frontierRow('func:callee-B', 'callee')]; + } + if (query.includes('COUNT(DISTINCT s.id)') || query.includes('RETURN s.id AS sid')) return []; + return [TARGET_ROW]; + }); + + const result = await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + line: 8, + }); + + expect(result.error).toBeUndefined(); + const items = result.interproceduralByDepth[1] as Array<{ id: string; pdgEvidence: string }>; + const byId = new Map(items.map((i) => [i.id, i.pdgEvidence])); + // Only the in-slice id is proven; the same-named sibling is a proof failure. + expect(byId.get('func:callee-A')).toBe('callgraph-bridge'); + expect(byId.get('func:callee-B')).toBe('unproven-bridge'); + }); + + it('drops the truncation sentinel from the slice id set (R7 — never matched as an id)', async () => { + vi.spyOn( + backend as unknown as { _runImpactPDG: () => Promise }, + '_runImpactPDG', + ).mockResolvedValueOnce({ ...PDG_SLICE_RESULT }); + + // calleeIds carries the sentinel '*' alongside a real id; callees has NO sentinel + // (so the names-sentinel short-circuit does not fire and the id path runs). A + // reached callee whose id is literally '*' must be UNPROVEN — the sentinel was + // filtered out of the id set, so it can never false-match. (TAB-joined per the + // CALLEE_ID_SEP delimiter — ids can contain spaces, so the cell is not space-joined.) + vi.mocked(executeParameterized).mockImplementation(async (_repo, query) => { + if (query.includes('RETURN b.calleeIds')) return [{ calleeIds: 'func:callee-A\t*' }]; + if (query.includes('RETURN b.callees')) return [{ callees: 'callee' }]; + if (query.includes('r.type IN $relTypes') && !query.includes('STEP_IN_PROCESS')) { + return [frontierRow('func:callee-A', 'callee'), frontierRow('*', 'callee')]; + } + if (query.includes('COUNT(DISTINCT s.id)') || query.includes('RETURN s.id AS sid')) return []; + return [TARGET_ROW]; + }); + + const result = await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + line: 8, + }); + + expect(result.error).toBeUndefined(); + const items = result.interproceduralByDepth[1] as Array<{ id: string; pdgEvidence: string }>; + const byId = new Map(items.map((i) => [i.id, i.pdgEvidence])); + expect(byId.get('func:callee-A')).toBe('callgraph-bridge'); + expect(byId.get('*')).toBe('unproven-bridge'); + }); +}); diff --git a/gitnexus/test/unit/calltool-dispatch.test.ts b/gitnexus/test/unit/calltool-dispatch.test.ts index 84bdc4e69..a3c501573 100644 --- a/gitnexus/test/unit/calltool-dispatch.test.ts +++ b/gitnexus/test/unit/calltool-dispatch.test.ts @@ -47,6 +47,14 @@ vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { listRegisteredRepos: vi.fn().mockResolvedValue([]), cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), findSiblingClones: vi.fn().mockResolvedValue([]), + // U2: expose loadMeta as a spy that delegates to the REAL implementation by + // default (so branch-scope resolution, #2106, is unaffected). The + // impact-mode block overrides it per-test to stamp a READY PDG layer, so the + // U2 layer-presence probe falls THROUGH to the post-check surface (the + // `_runImpactPDG` delegate / ambiguous fan-out) those tests assert. The + // four-state degradation contract itself is covered in + // test/integration/impact-pdg-degradation.test.ts. + loadMeta: vi.fn(actual.loadMeta), }; }); @@ -97,7 +105,16 @@ import { REPO_ID_HASH_LENGTH, parseListReposPagination, } from '../../src/mcp/local/local-backend.js'; -import { listRegisteredRepos, cleanupOldKuzuFiles } from '../../src/storage/repo-manager.js'; +import { + betterBridgeEvidence, + pdgBridgeEvidenceForImpact, +} from '../../src/mcp/local/pdg-impact.js'; +import { CALLEES_TRUNCATED_SENTINEL } from '../../src/core/ingestion/cfg/emit.js'; +import { + listRegisteredRepos, + cleanupOldKuzuFiles, + loadMeta, +} from '../../src/storage/repo-manager.js'; import { getGitRoot } from '../../src/storage/git.js'; import { _captureLogger } from '../../src/core/logger.js'; import { @@ -1385,6 +1402,535 @@ describe('LocalBackend.callTool', () => { }); }); +// ─── impact mode param (KTD1/KTD5/KTD12 — U1) ─────────────────────── +// +// The MCP JSON-schema enum is advisory only (server forwards args +// unvalidated, callTool is reachable directly), so the backend `mode` +// validation is load-bearing. These tests pin: callgraph is the unchanged +// default, pdg routes to the extracted traversal plus interprocedural symbol +// reach, invalid modes hard-error, and the remaining incompatible params / +// @group targets are rejected. + +describe('LocalBackend impact mode (KTD1/KTD5/KTD12)', () => { + let backend: LocalBackend; + + // Resolve the target to a single Function so impact reaches the single-branch + // dispatch (callgraph BFS or the PDG traversal). The callgraph BFS then issues + // executeQuery for its frontier; the PDG path delegates to runImpactPDG. + function resolveSingleTarget() { + (executeParameterized as any).mockResolvedValue([ + { id: 'func:main', name: 'main', type: 'Function', filePath: 'src/index.ts' }, + ]); + (executeQuery as any).mockResolvedValue([]); + } + + beforeEach(async () => { + vi.clearAllMocks(); + platformMocks.isVectorExtensionSupportedByPlatform.mockReturnValue(true); + // U2: stamp a READY PDG layer (both caps) so the layer-presence probe in + // `_impactImpl` falls THROUGH to the mode-dispatch surface these tests pin + // (the `_runImpactPDG` delegate / the ambiguous fan-out under `mode:'pdg'`). + // Degraded-layer behavior is owned by the integration degradation suite. + vi.mocked(loadMeta).mockResolvedValue({ + pdg: { maxCdgEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }, + } as any); + backend = new LocalBackend(); + setupSingleRepo(); + await backend.init(); + }); + + it('mode absent → callgraph result (target populated, no mode-error, BFS runs)', async () => { + resolveSingleTarget(); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + const result = await backend.callTool('impact', { target: 'main', direction: 'upstream' }); + // A clean callgraph result carries no mode error and runs the BFS. + expect(result.error ?? '').not.toMatch(/Invalid "mode"/); + expect(result.error ?? '').not.toMatch(/not yet implemented/); + expect(result.target).toBeDefined(); + expect(bfsSpy).toHaveBeenCalledTimes(1); + }); + + it("mode:'callgraph' and mode:undefined are byte-identical to absent (regression guard)", async () => { + resolveSingleTarget(); + const absent = await backend.callTool('impact', { target: 'main', direction: 'upstream' }); + const callgraph = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: 'callgraph', + }); + const undef = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: undefined, + }); + expect(callgraph).toEqual(absent); + expect(undef).toEqual(absent); + }); + + it("mode:'pdg' routes to the PDG traversal and attaches interprocedural symbol reach", async () => { + resolveSingleTarget(); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: 'pdg', + }); + // The call reaches the real `_runImpactPDG` traversal, then composes the + // interprocedural symbol reach into the same pdg result. + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(Array.isArray(result.reachableBlocks)).toBe(true); + expect(result.pdgInterprocedural).toBeDefined(); + expect(bfsSpy).toHaveBeenCalledTimes(1); + }); + + it("mode:'pdg' labels interprocedural symbols as a callgraph bridge", async () => { + resolveSingleTarget(); + vi.spyOn(backend as any, '_runImpactBFS').mockResolvedValueOnce({ + target: { id: 'func:main', name: 'main', type: 'Function', filePath: 'src/index.ts' }, + direction: 'downstream', + impactedCount: 1, + risk: 'LOW', + summary: { direct: 1, processes_affected: 0, modules_affected: 0 }, + byDepthCounts: { 1: 1 }, + affected_processes: [], + affected_modules: [], + byDepth: { + 1: [ + { + depth: 1, + id: 'func:callee', + name: 'callee', + type: 'Function', + filePath: 'src/callee.ts', + }, + ], + }, + }); + + const result = await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + }); + + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(result.pdgInterprocedural.evidence).toBe('callgraph-bridge'); + expect(result.pdgInterprocedural.evidenceCounts['callgraph-bridge']).toBe(1); + expect(result.pdgEvidence.interprocedural).toBe('callgraph-bridge'); + expect(result.interproceduralByDepth[1][0].pdgEvidence).toBe('callgraph-bridge'); + expect(result.note).toContain('labeled as a PDG evidence bridge'); + }); + + it("mode:'pdg' preserves unproven bridge evidence when call-site proof is unavailable", async () => { + resolveSingleTarget(); + vi.spyOn(backend as any, '_runImpactBFS').mockResolvedValueOnce({ + target: { id: 'func:main', name: 'main', type: 'Function', filePath: 'src/index.ts' }, + direction: 'downstream', + impactedCount: 1, + risk: 'LOW', + summary: { direct: 1, processes_affected: 0, modules_affected: 0 }, + byDepthCounts: { 1: 1 }, + affected_processes: [], + affected_modules: [], + byDepth: { + 1: [ + { + depth: 1, + id: 'func:callee', + name: 'callee', + type: 'Function', + filePath: 'src/callee.ts', + pdgEvidence: 'unproven-bridge', + }, + ], + }, + }); + + const result = await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + }); + + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(result.pdgInterprocedural.evidence).toBe('unproven-bridge'); + expect(result.pdgInterprocedural.evidenceCounts['unproven-bridge']).toBe(1); + expect(result.pdgEvidence.interprocedural).toBe('unproven-bridge'); + expect(result.note).toContain('labeled unproven-bridge'); + }); + + it.each([['PDG'], ['pgd'], [''], [0], [null]])( + 'invalid mode %j → structured {error}, never a callgraph result (KTD5 anti-silent-fallback)', + async (bad) => { + resolveSingleTarget(); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: bad as any, + }); + expect(result.error).toMatch(/Invalid "mode"/); + expect(result.risk).toBe('UNKNOWN'); + // A typo'd mode must NEVER quietly run callgraph. + expect(bfsSpy).not.toHaveBeenCalled(); + }, + ); + + it.each([['callgraph'], [undefined]])( + 'line param with mode:%j → structured {error} (line is PDG-only), never a callgraph result', + async (mode) => { + resolveSingleTarget(); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: mode as any, + line: 8, + }); + expect(result.error).toMatch(/'line' is only supported with mode:'pdg'/); + expect(result.risk).toBe('UNKNOWN'); + // A PDG-only param on the callgraph path must NOT silently run the BFS. + expect(bfsSpy).not.toHaveBeenCalled(); + }, + ); + + it.each([[0], [-1], [1.5]])( + "mode:'pdg' + non-positive-integer line %j → structured {error}, never routed to traversal", + async (badLine) => { + resolveSingleTarget(); + const pdgSpy = vi.spyOn(backend as any, '_runImpactPDG'); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: 'pdg', + line: badLine as any, + }); + expect(result.error).toMatch(/'line' must be a positive integer/); + expect(result.risk).toBe('UNKNOWN'); + // The validation fires BEFORE the traversal — a bad line never seeds a slice. + expect(pdgSpy).not.toHaveBeenCalled(); + }, + ); + + it("mode:'pdg' + downstream line:8 routes to the PDG traversal and seeds bridge evidence", async () => { + resolveSingleTarget(); + // The target-resolution row doubles as the calleesOfBlocks row: `callees` + // ('callee') is the leaf name persisted on the slice's BasicBlock, the + // statement-precise substrate the bridge keys on. + (executeParameterized as any).mockResolvedValue([ + { + id: 'func:main', + name: 'main', + type: 'Function', + filePath: 'src/index.ts', + callees: 'callee', + }, + ]); + // A line-seeded downstream slice with one reachable block → the dispatch + // queries that block's callees and seeds the bridge with them. + const pdgSpy = vi.spyOn(backend as any, '_runImpactPDG').mockResolvedValueOnce({ + mode: 'pdg', + target: { id: 'func:main', name: 'main', type: 'Function', filePath: 'src/index.ts' }, + direction: 'downstream', + risk: 'UNKNOWN', + impactedCount: 0, + epistemic: 'pdg-intra-procedural', + reachableBlocks: ['BasicBlock:src/index.ts:8:0:1'], + // Intra-only slice (no inter-procedural hop) ⇒ the intra reach the bridge + // keys on equals reachableBlocks (FIX 6: bridge keys on intraReachableBlocks). + intraReachableBlocks: ['BasicBlock:src/index.ts:8:0:1'], + blockCount: 1, + affectedStatements: [{ line: 8, filePath: 'src/index.ts', text: 'callee()' }], + affectedStatementCount: 1, + criterionLine: 8, + }); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + line: 8, + }); + // A valid line routes cleanly into the PDG engine — no line/mode error. + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(pdgSpy).toHaveBeenCalledTimes(1); + expect(bfsSpy).toHaveBeenCalledTimes(1); + const bridge = bfsSpy.mock.calls[0][4].pdgBridge; + // The bridge now carries the slice's callee names (statement-precise reach), + // resolved from BasicBlock.callees — not the dead call-site-line keys. + expect(bridge).toBeDefined(); + expect([...bridge.sliceCalleeNames]).toContain('callee'); + expect(result.pdgInterprocedural).toBeDefined(); + }); + + it("mode:'pdg' downstream: a callee invoked ON the seeded line is proven even with no downstream dependents", async () => { + // Regression for the PR #2227 tri-review P2: the seed block is excluded from + // `reachableBlocks` (seed-minus-reachable convention), so a callee called + // directly on the changed line — with NO downstream-dependent block — used to + // be dropped from the statement-precise set. The dispatch now unions the seed + // block's callees, so it must be proven. + resolveSingleTarget(); + (executeParameterized as any).mockResolvedValue([ + { + id: 'func:main', + name: 'main', + type: 'Function', + filePath: 'src/index.ts', + callees: 'seedCallee', + }, + ]); + // reachableBlocks EMPTY (line N has no downstream dependents) but seedBlocks + // carries the changed line's own block — the case that regressed. + vi.spyOn(backend as any, '_runImpactPDG').mockResolvedValueOnce({ + mode: 'pdg', + target: { id: 'func:main', name: 'main', type: 'Function', filePath: 'src/index.ts' }, + direction: 'downstream', + risk: 'UNKNOWN', + impactedCount: 0, + epistemic: 'pdg-intra-procedural', + reachableBlocks: [], + intraReachableBlocks: [], + seedBlocks: ['BasicBlock:src/index.ts:8:0:0'], + blockCount: 0, + affectedStatements: [], + affectedStatementCount: 0, + criterionLine: 8, + }); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + line: 8, + }); + const bridge = bfsSpy.mock.calls[0][4].pdgBridge; + // The bridge is seeded from the seed block (not just reachableBlocks), so the + // seed-line callee is provable. + expect(bridge).toBeDefined(); + expect([...bridge.sliceCalleeNames]).toContain('seedCallee'); + }); + + it('betterBridgeEvidence keeps callgraph-bridge regardless of parent order (U3 order-independence)', () => { + const proven = { evidence: 'callgraph-bridge' as const, basis: 'in slice' }; + const unproven = { evidence: 'unproven-bridge' as const, basis: 'not in slice' }; + // A node reached from a proven and an unproven parent is proven either way — + // the diamond label does not depend on which edge the BFS visits first. + expect(betterBridgeEvidence(unproven, proven).evidence).toBe('callgraph-bridge'); + expect(betterBridgeEvidence(proven, unproven).evidence).toBe('callgraph-bridge'); + // First verdict wins when neither is stronger; undefined existing takes the candidate. + expect(betterBridgeEvidence(undefined, unproven).evidence).toBe('unproven-bridge'); + expect(betterBridgeEvidence(unproven, unproven).evidence).toBe('unproven-bridge'); + }); + + it('pdgBridgeEvidenceForImpact treats a truncated-slice (sentinel) as callee-unknown → proven', () => { + // A slice block that hit the per-statement site cap has an incomplete callee + // list; the sentinel forces callgraph-equal so an absent-but-real callee is + // not under-proven. + const truncated = pdgBridgeEvidenceForImpact({ + bridge: { + sliceCalleeNames: new Set([CALLEES_TRUNCATED_SENTINEL, 'foo']), + sliceCalleeIds: new Set(), + }, + depth: 1, + calleeName: 'unrelatedNotInSlice', + }); + expect(truncated.evidence).toBe('callgraph-bridge'); + // Without the sentinel, a callee not in the slice is unproven. + const notTruncated = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeNames: new Set(['foo']), sliceCalleeIds: new Set() }, + depth: 1, + calleeName: 'unrelatedNotInSlice', + }); + expect(notTruncated.evidence).toBe('unproven-bridge'); + }); + + it("mode:'pdg' degrades gracefully when the slice-callees query fails (no bridge, no throw)", async () => { + // calleesOfBlocks swallows a DB error and returns an empty set, so the bridge + // is not built and the inter-procedural reach falls back to callgraph-equal — + // never surfacing the error or producing a partial proven/unproven labeling. + resolveSingleTarget(); + // The slice-callees query (RETURN b.callees) throws; every other query (target + // resolution) returns the resolved symbol row. + vi.mocked(executeParameterized).mockImplementation(async (_repo, query) => { + if (query.includes('RETURN b.callees')) throw new Error('slice-callees query failed'); + return [{ id: 'func:main', name: 'main', type: 'Function', filePath: 'src/index.ts' }]; + }); + // A line-seeded downstream slice so calleesOfBlocks is attempted. + vi.spyOn(backend as any, '_runImpactPDG').mockResolvedValueOnce({ + mode: 'pdg', + target: { id: 'func:main', name: 'main', type: 'Function', filePath: 'src/index.ts' }, + direction: 'downstream', + risk: 'UNKNOWN', + impactedCount: 0, + epistemic: 'pdg-intra-procedural', + reachableBlocks: ['BasicBlock:src/index.ts:8:0:1'], + intraReachableBlocks: ['BasicBlock:src/index.ts:8:0:1'], + seedBlocks: ['BasicBlock:src/index.ts:8:0:0'], + blockCount: 1, + affectedStatements: [{ line: 8, filePath: 'src/index.ts', text: 'callee()' }], + affectedStatementCount: 1, + criterionLine: 8, + }); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + line: 8, + }); + // The error was swallowed: no bridge passed to the BFS, and no error surfaced. + expect(result.error).toBeUndefined(); + expect(bfsSpy.mock.calls[0][4].pdgBridge).toBeUndefined(); + }); + + it("mode:'pdg' + crossDepth → hard {error} (single-repo PDG impact)", async () => { + resolveSingleTarget(); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: 'pdg', + crossDepth: 2, + }); + expect(result.error).toMatch(/not supported with mode:'pdg'/); + expect(result.error).toContain('crossDepth'); + expect(bfsSpy).not.toHaveBeenCalled(); + }); + + it.each([ + ['relationTypes', { relationTypes: ['CALLS'] }, (opts: any) => opts.relationTypes], + ['minConfidence', { minConfidence: 0.5 }, (opts: any) => opts.minConfidence], + ])("mode:'pdg' + %s feeds the interprocedural symbol reach", async (_label, extra, readOpt) => { + resolveSingleTarget(); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS').mockResolvedValueOnce({ + target: { id: 'func:main', name: 'main', type: 'Function', filePath: 'src/index.ts' }, + direction: 'upstream', + impactedCount: 0, + risk: 'LOW', + summary: { direct: 0, processes_affected: 0, modules_affected: 0 }, + byDepthCounts: {}, + affected_processes: [], + affected_modules: [], + byDepth: {}, + }); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: 'pdg', + ...extra, + }); + expect(result.error).toBeUndefined(); + expect(result.mode).toBe('pdg'); + expect(bfsSpy).toHaveBeenCalledTimes(1); + expect(readOpt(bfsSpy.mock.calls[0][4])).toBeDefined(); + }); + + it("ambiguous target under mode:'pdg' never invokes interprocedural fan-out (KTD5 ambiguous trap)", async () => { + // Two same-name Functions → resolver returns ambiguous. + (executeParameterized as any).mockResolvedValue([ + { + id: 'func:login:1', + name: 'login', + type: 'Function', + filePath: 'src/auth.ts', + startLine: 5, + }, + { + id: 'func:login:2', + name: 'login', + type: 'Function', + filePath: 'src/admin/login.ts', + startLine: 8, + }, + ]); + const bfsSpy = vi.spyOn(backend as any, '_runImpactBFS'); + const result = await backend.callTool('impact', { + target: 'login', + direction: 'upstream', + mode: 'pdg', + }); + expect(result.status).toBe('ambiguous'); + expect(result.mode).toBe('pdg'); + expect(result.candidates).toHaveLength(2); + expect(result.impactedCount).toBe(0); + expect(result.risk).toBe('UNKNOWN'); + // The callgraph per-candidate probe fan-out MUST NOT run under pdg. + expect(bfsSpy).not.toHaveBeenCalled(); + // No per-candidate blast radius is computed yet (U4), so the candidate + // entries carry no impactedCount field from a callgraph probe. + for (const c of result.candidates) { + expect(c.impactedCount).toBeUndefined(); + } + }); + + it("unknown target with mode:'pdg' returns the normalized PDG error envelope", async () => { + (executeParameterized as any).mockResolvedValue([]); + const result = await backend.callTool('impact', { + target: 'missingSymbol', + direction: 'upstream', + mode: 'pdg', + }); + expect(result.error).toMatch(/not found/); + expect(result.mode).toBe('pdg'); + expect(result.target).toEqual({ name: 'missingSymbol' }); + expect(result.direction).toBe('upstream'); + expect(result.impactedCount).toBe(0); + expect(result.risk).toBe('UNKNOWN'); + }); + + it("runtime failures with mode:'pdg' return the normalized PDG error envelope", async () => { + const failing = new Error('pdg query failed'); + const implSpy = vi.spyOn(backend as any, '_impactImpl').mockRejectedValueOnce(failing); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'downstream', + mode: 'pdg', + }); + expect(result.error).toBe('pdg query failed'); + expect(result.mode).toBe('pdg'); + expect(result.target).toEqual({ name: 'main' }); + expect(result.direction).toBe('downstream'); + expect(result.impactedCount).toBe(0); + expect(result.risk).toBe('UNKNOWN'); + expect(result.suggestion).toMatch(/context/); + implSpy.mockRestore(); + }); + + it("@group target with mode:'pdg' is rejected (KTD12 — PDG is single-repo)", async () => { + resolveAtMemberMock.mockResolvedValue({ ok: true, repoPath: '/tmp/test-project' }); + const result = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: 'pdg', + repo: '@grp', + }); + expect(result.error).toMatch(/not supported for @group targets/); + expect(result.mode).toBe('pdg'); + expect(result.target).toEqual({ name: 'main' }); + expect(result.direction).toBe('upstream'); + expect(result.impactedCount).toBe(0); + expect(result.risk).toBe('UNKNOWN'); + }); + + it("@group target with mode:'callgraph' still forwards to group impact (unchanged)", async () => { + resolveAtMemberMock.mockResolvedValue({ ok: true, repoPath: '/tmp/test-project' }); + // groupImpact is reached only if the mode gate passes; we don't assert its + // payload (group infra is stubbed), only that no mode-error short-circuited. + const result = await backend.callTool('impact', { + target: 'main', + direction: 'upstream', + mode: 'callgraph', + repo: '@grp', + }); + expect(result?.error ?? '').not.toMatch(/not supported for @group targets/); + expect(result?.error ?? '').not.toMatch(/Invalid "mode"/); + }); +}); + // ─── Repo resolution ──────────────────────────────────────────────── describe('LocalBackend.resolveRepo', () => { diff --git a/gitnexus/test/unit/cfg-callee-ids-of-block.test.ts b/gitnexus/test/unit/cfg-callee-ids-of-block.test.ts new file mode 100644 index 000000000..839c848b3 --- /dev/null +++ b/gitnexus/test/unit/cfg-callee-ids-of-block.test.ts @@ -0,0 +1,336 @@ +import { describe, expect, it } from 'vitest'; +import { + CALLEES_TRUNCATED_SENTINEL, + CALLEE_ID_SEP, + calleeIdsOfBlock, + calleesOfBlock, + emitFileCfgs, +} from '../../src/core/ingestion/cfg/emit.js'; +import { DEFAULT_PDG_MAX_SITES_PER_STATEMENT } from '../../src/core/ingestion/cfg/visitors/call-site-harvest.js'; +import { calleeIdPosKey } from '../../src/core/ingestion/scope-resolution/graph-bridge/callee-id-sink.js'; +import { splitCalleeIds } from '../../src/mcp/local/pdg-impact.js'; +import { cfgOf } from '../helpers/ts-cfg-harness.js'; +import { allSites } from '../helpers/cfg-harness.js'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import type { + BasicBlockData, + FunctionCfg, + SiteRecord, +} from '../../src/core/ingestion/cfg/types.js'; +import type { GraphNode } from 'gitnexus-shared'; + +/** + * U3 (#2227 follow-up plan) — `BasicBlock.calleeIds` via the exact call-site + * position join. + * + * `calleeIdsOfBlock(block, fileMap)` mirrors {@link calleesOfBlock}, but joins + * each site's U1 `at` anchor to U2's resolved-id map (`posKey → Set`) + * instead of slicing the leaf name. The CHARACTERIZATION block below proves the + * KTD7 round-trip: a synthetic map keyed at the EXACT `at` positions the real + * harvester produced resolves back to exactly the block's own callees — i.e. + * U1's `at` and U2's `calleeIdPosKey` agree on the coordinate. + */ + +// ── pure-helper scenarios (hand-built blocks, mirroring cfg-callees-of-block) ── + +const callSite = (callee: string, at: readonly [number, number]): SiteRecord => ({ + kind: 'call', + callee, + at, +}); + +const block = (statements: BasicBlockData['statements']): BasicBlockData => ({ + index: 0, + startLine: 1, + endLine: 1, + text: '', + kind: 'normal', + statements, +}); + +describe('calleeIdsOfBlock', () => { + it('emits sorted, de-duplicated resolved ids joined by exact site position', () => { + const s1 = callSite('foo', [1, 0]); + const s2 = callSite('bar', [2, 4]); + const fileMap = new Map>([ + [calleeIdPosKey(1, 0), new Set(['idA'])], + [calleeIdPosKey(2, 4), new Set(['idB'])], + ]); + // idB then idA in source order, but the output is SORTED (mirrors callees). + const result = calleeIdsOfBlock( + block([{ line: 1, defs: [], uses: [], sites: [s2, s1] }]), + fileMap, + ); + expect(result).toBe('idA\tidB'); + }); + + it('unions multi-target dispatch ids at one position (KTD8/R2)', () => { + const fileMap = new Map>([ + [calleeIdPosKey(3, 2), new Set(['idX', 'idY'])], + ]); + const result = calleeIdsOfBlock( + block([{ line: 3, defs: [], uses: [], sites: [callSite('dispatch', [3, 2])] }]), + fileMap, + ); + expect(result).toBe('idX\tidY'); + }); + + it('a resolved id containing a space round-trips through calleeIdsOfBlock + splitCalleeIds (#2227)', () => { + // C++ overload ids embed multi-word primitives (`unsigned char`) and file + // paths can contain spaces, so a resolved id can legitimately hold a space. + // The TAB delimiter keeps it in ONE field; a space-join would fragment it + // and silently drop inter-procedural reach to that callee. + const spaceId = 'Method:src/my file.cpp:S::f~shape:unsigned char:none:pointer:1'; + const fileMap = new Map>([ + [calleeIdPosKey(1, 0), new Set([spaceId])], + [calleeIdPosKey(2, 4), new Set(['idPlain'])], + ]); + const cell = calleeIdsOfBlock( + block([ + { line: 1, defs: [], uses: [], sites: [callSite('f', [1, 0]), callSite('g', [2, 4])] }, + ]), + fileMap, + ); + // Tab-joined: the space-id keeps its internal spaces inside one field. + expect(cell).toContain(CALLEE_ID_SEP); + expect(cell).toContain('unsigned char'); + // splitCalleeIds recovers BOTH ids WHOLE — the space-id is not fragmented. + expect([...splitCalleeIds(cell)].sort()).toEqual([spaceId, 'idPlain'].sort()); + }); + + it('carries the resolved id even when the leaf name differs from the call-site leaf (alias, R1)', () => { + // The site leaf is `aliased`, but the resolution map binds its position to + // the canonical symbol id `id:realFn` — `calleeIds` carries the RESOLVED id, + // not the syntactic alias. + const fileMap = new Map>([ + [calleeIdPosKey(5, 8), new Set(['id:realFn'])], + ]); + const result = calleeIdsOfBlock( + block([{ line: 5, defs: [], uses: [], sites: [callSite('aliased', [5, 8])] }]), + fileMap, + ); + expect(result).toBe('id:realFn'); + }); + + it('skips member-read sites and sites whose position is not in the map', () => { + const fileMap = new Map>([ + [calleeIdPosKey(1, 4), new Set(['idHit'])], + ]); + const result = calleeIdsOfBlock( + block([ + { + line: 1, + defs: [], + uses: [], + sites: [ + { kind: 'member-read', property: 'body', at: [1, 4] }, // member-read → skipped + callSite('hit', [1, 4]), // position in map → idHit + callSite('miss', [9, 9]), // position absent from map → no id + ], + }, + ]), + fileMap, + ); + expect(result).toBe('idHit'); + }); + + it('flags the block callee-unknown with the sentinel when a statement hits the site cap (R7)', () => { + const cappedSites: SiteRecord[] = Array.from( + { length: DEFAULT_PDG_MAX_SITES_PER_STATEMENT }, + (_unused, i) => callSite('foo', [1, i]), + ); + const fileMap = new Map>([ + [calleeIdPosKey(1, 0), new Set(['idFoo'])], + ]); + const result = calleeIdsOfBlock( + block([{ line: 1, defs: [], uses: [], sites: cappedSites }]), + fileMap, + ); + // The sentinel sorts first ('*' < letters) and rides alongside the real ids. + expect(result.split(CALLEE_ID_SEP)).toContain(CALLEES_TRUNCATED_SENTINEL); + expect(result.split(CALLEE_ID_SEP)).toContain('idFoo'); + }); + + it('returns an empty string when the map is absent (pdg off / degraded — R3)', () => { + const b = block([{ line: 1, defs: [], uses: [], sites: [callSite('foo', [1, 0])] }]); + expect(calleeIdsOfBlock(b, undefined)).toBe(''); + // callees is unaffected — the leaf-name fallback substrate is still present. + expect(calleesOfBlock(b)).toBe('foo'); + }); + + it('returns an empty string for a block with no call sites', () => { + const fileMap = new Map>([ + [calleeIdPosKey(1, 0), new Set(['idA'])], + ]); + expect(calleeIdsOfBlock(block([{ line: 1, defs: [], uses: [] }]), fileMap)).toBe(''); + expect(calleeIdsOfBlock(block(undefined), fileMap)).toBe(''); + }); +}); + +// ── emitFileCfgs property wiring ───────────────────────────────────────────── + +/** The emitted BasicBlock node properties, keyed by block index, for a 1-fn cfg. */ +function emittedBlockProps( + cfg: FunctionCfg, + calleeIdMap?: ReadonlyMap>, +): Map { + const graph = createKnowledgeGraph(); + emitFileCfgs(graph, [cfg], undefined, undefined, calleeIdMap); + const out = new Map(); + for (let i = 0; i < cfg.blocks.length; i++) { + const id = `BasicBlock:${cfg.filePath}:${cfg.functionStartLine}:${cfg.functionStartColumn}:${i}`; + const node = graph.getNode(id); + expect(node).toBeDefined(); + out.set(i, (node as GraphNode).properties); + } + return out; +} + +describe('emitFileCfgs — calleeIds property', () => { + it('emits calleeIds = "" for every block when no map is passed (pdg off, R4)', () => { + const cfg = cfgOf(`function f(arr) { arr.map(x => foo(x)); bar(); }`); + const props = emittedBlockProps(cfg); + for (const p of props.values()) { + expect(p).toMatchObject({ calleeIds: '' }); + } + }); + + it('emits a non-empty calleeIds joined from the map, alongside the unchanged callees', () => { + const cfg = cfgOf(`function f(arr) { arr.map(x => foo(x)); bar(); }`); + // Synthetic resolved-id map keyed at the EXACT positions the harvester + // produced for `arr.map` and `bar`. + const fileMap = new Map>([ + [calleeIdPosKey(1, 18), new Set(['id:map'])], // arr.map + [calleeIdPosKey(1, 40), new Set(['id:bar'])], // bar + ]); + const props = emittedBlockProps(cfg, fileMap); + const callBlock = [...props.values()].find( + (p) => typeof p.calleeIds === 'string' && p.calleeIds.length > 0, + ); + expect(callBlock).toMatchObject({ + callees: 'bar map', // leaf names, sorted — UNCHANGED by the id wiring + calleeIds: 'id:bar\tid:map', // resolved ids, sorted (TAB-joined, CALLEE_ID_SEP) + }); + }); +}); + +// ── CHARACTERIZATION-FIRST (KTD7) — the at ↔ calleeIdPosKey round-trip ──────── + +/** + * Build a synthetic `posKey → Set` map keyed at the EXACT `at` + * positions the produced SiteRecords carry, assigning each call/new site a + * synthetic resolved id `id:`. This is what U2 would have captured had + * the CALLS resolution keyed `atRange` on the same anchor. The id round-trips + * back to its callee via {@link idToCallee}. + */ +function syntheticMapFromSites(cfg: FunctionCfg): { + fileMap: ReadonlyMap>; + idToCallee: Map; +} { + const fileMap = new Map>(); + const idToCallee = new Map(); + for (const site of allSites(cfg)) { + // Only call/new sites carry `at`; the join consumes exactly those. + const at = site.at; + const callee = site.callee; + if (at === undefined || callee === undefined) continue; + const id = `id:${callee}`; + idToCallee.set(id, callee.slice(callee.lastIndexOf('.') + 1)); + const key = calleeIdPosKey(at[0], at[1]); + const set = fileMap.get(key) ?? new Set(); + set.add(id); + fileMap.set(key, set); + } + return { fileMap, idToCallee }; +} + +/** Leaf names of `calleeIds`, recovered through the synthetic id→callee map. */ +function calleeNamesViaIds(calleeIds: string, idToCallee: Map): string { + const names = calleeIds + .split(CALLEE_ID_SEP) + .filter((tok) => tok.length > 0) + .map((id) => { + const name = idToCallee.get(id); + expect(name).toBeDefined(); + return name as string; + }); + return [...new Set(names)].sort().join(' '); +} + +/** + * For every block, the id set (mapped back to callee leaf names) must equal the + * block's OWN `calleesOfBlock` names — proving the `at` positions land on the + * right sites. This is the per-block KTD7 round-trip assertion. + */ +function assertRoundTrip(cfg: FunctionCfg): void { + const { fileMap, idToCallee } = syntheticMapFromSites(cfg); + for (const b of cfg.blocks) { + const calleeIds = calleeIdsOfBlock(b, fileMap); + const names = calleesOfBlock(b); + // The id set, resolved back through the synthetic map, must reproduce the + // block's own leaf-name set EXACTLY (no divergence): same positions, same + // partitioning. (No aliases in these fixtures, so "modulo aliases" is exact + // equality.) + expect(calleeNamesViaIds(calleeIds, idToCallee)).toBe(names); + } +} + +describe('KTD7 characterization — at ↔ calleeIdPosKey round-trip', () => { + it('round-trips a multi-line statement with an argument-position call on the next line', () => { + // The inner `inner(a)` call begins on line 3, the statement head on line 2. + // A statement-line join would mis-attribute inner's id — the position join + // must land it on its own line. + const cfg = cfgOf(`function f(a, b) {\n outer(\n inner(a)\n );\n}`); + assertRoundTrip(cfg); + // Pin the produced positions so a future anchor regression is caught here. + const outer = allSites(cfg).find((s) => s.callee === 'outer'); + const inner = allSites(cfg).find((s) => s.callee === 'inner'); + expect(outer).toMatchObject({ at: [2, 2] }); + expect(inner).toMatchObject({ at: [3, 4] }); + }); + + it('round-trips a member chain alongside a member-read site (member-read carries no id)', () => { + const cfg = cfgOf(`function f(a, x) { a.b.c(x); svc.run(); }`); + assertRoundTrip(cfg); + // The member-read site (no `at`) must not contribute an id. + const memberReads = allSites(cfg).filter((s) => s.kind === 'member-read'); + expect(memberReads.length).toBeGreaterThan(0); + expect(memberReads.every((s) => s.at === undefined)).toBe(true); + }); + + it('single-line closure: calleeIds carries the map + bar ids but NOT the nested foo (the bug the position join fixes)', () => { + // `arr.map(x => foo(x)); bar();` — `foo` is a nested-fn site the harvester + // EXCLUDES, so it produces no top-level SiteRecord (no `at`). The position + // join therefore cannot include foo's id even if one existed at foo's + // position. We assert this directly: a map that ALSO binds foo's source + // position (col 30) must NOT leak foo into the outer block's calleeIds. + const cfg = cfgOf(`function f(arr) { arr.map(x => foo(x)); bar(); }`); + + // The block's real call sites are exactly [arr.map@[1,18], bar@[1,40]]. + const sites = allSites(cfg).filter((s) => s.kind === 'call' || s.kind === 'new'); + expect(sites.map((s) => s.callee)).toEqual(['arr.map', 'bar']); + + // Adversarial map: binds the real two positions to map/bar ids AND binds + // foo's source column (30) to a foo id. Because no SiteRecord carries + // foo's position, the join cannot reach it. + const fileMap = new Map>([ + [calleeIdPosKey(1, 18), new Set(['id:map'])], + [calleeIdPosKey(1, 40), new Set(['id:bar'])], + [calleeIdPosKey(1, 30), new Set(['id:foo'])], // foo's position — never joined + ]); + + // The single call-bearing block. + const callBlock = cfg.blocks.find((b) => + (b.statements ?? []).some((s) => (s.sites ?? []).some((x) => x.kind === 'call')), + ); + expect(callBlock).toBeDefined(); + const calleeIds = calleeIdsOfBlock(callBlock as BasicBlockData, fileMap); + + const idSet = new Set(calleeIds.split(CALLEE_ID_SEP).filter((t) => t.length > 0)); + expect(idSet.has('id:map')).toBe(true); + expect(idSet.has('id:bar')).toBe(true); + expect(idSet.has('id:foo')).toBe(false); + // Sorted, exact set (TAB-joined, CALLEE_ID_SEP). + expect(calleeIds).toBe('id:bar\tid:map'); + }); +}); diff --git a/gitnexus/test/unit/cfg-callees-of-block.test.ts b/gitnexus/test/unit/cfg-callees-of-block.test.ts new file mode 100644 index 000000000..4135fa07c --- /dev/null +++ b/gitnexus/test/unit/cfg-callees-of-block.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from 'vitest'; +import { CALLEES_TRUNCATED_SENTINEL, calleesOfBlock } from '../../src/core/ingestion/cfg/emit.js'; +import { DEFAULT_PDG_MAX_SITES_PER_STATEMENT } from '../../src/core/ingestion/cfg/visitors/call-site-harvest.js'; +import type { BasicBlockData, SiteRecord } from '../../src/core/ingestion/cfg/types.js'; + +const callSite = (callee: string): SiteRecord => ({ kind: 'call', callee }); + +const block = (statements: BasicBlockData['statements']): BasicBlockData => ({ + index: 0, + startLine: 1, + endLine: 1, + text: '', + kind: 'normal', + statements, +}); + +describe('calleesOfBlock', () => { + it('emits sorted, de-duplicated leaf callee names (dotted paths reduced to the leaf)', () => { + const result = calleesOfBlock( + block([ + { line: 1, defs: [], uses: [], sites: [callSite('child_process.exec'), callSite('foo')] }, + { line: 2, defs: [], uses: [], sites: [callSite('a.b.bar'), callSite('foo')] }, + ]), + ); + expect(result).toBe('bar exec foo'); + }); + + it('ignores member-read sites and sites without a callee', () => { + const result = calleesOfBlock( + block([ + { + line: 1, + defs: [], + uses: [], + sites: [{ kind: 'member-read', property: 'body' }, { kind: 'call' }, callSite('only')], + }, + ]), + ); + expect(result).toBe('only'); + }); + + it('flags a block callee-unknown with the sentinel when a statement hits the site cap', () => { + const cappedSites: SiteRecord[] = Array.from( + { length: DEFAULT_PDG_MAX_SITES_PER_STATEMENT }, + () => callSite('foo'), + ); + const result = calleesOfBlock(block([{ line: 1, defs: [], uses: [], sites: cappedSites }])); + // The sentinel sorts first ('*' < letters) and rides alongside the real names. + expect(result.split(' ')).toContain(CALLEES_TRUNCATED_SENTINEL); + expect(result.split(' ')).toContain('foo'); + }); + + it('returns an empty string for a block with no call sites', () => { + expect(calleesOfBlock(block([{ line: 1, defs: [], uses: [] }]))).toBe(''); + expect(calleesOfBlock(block(undefined))).toBe(''); + }); +}); diff --git a/gitnexus/test/unit/cfg-site-position.test.ts b/gitnexus/test/unit/cfg-site-position.test.ts new file mode 100644 index 000000000..bbab85213 --- /dev/null +++ b/gitnexus/test/unit/cfg-site-position.test.ts @@ -0,0 +1,124 @@ +import { describe, it, expect } from 'vitest'; +import type { FunctionCfg, SiteRecord } from '../../src/core/ingestion/cfg/types.js'; +import { cfgOf } from '../helpers/ts-cfg-harness.js'; +import { allSites } from '../helpers/cfg-harness.js'; + +/** + * U1 (#2227 follow-up plan) — `SiteRecord.at` call-site anchor position. + * + * Each call/new `SiteRecord` is stamped with `at: [line (1-based), col + * (0-based)]`, recorded by the worker harvester at the call/new node where it + * reads the callee. A downstream unit joins each site to its resolved callee id + * by EXACT position, so `at` MUST be the SAME anchor the CALLS-edge resolution + * keys on (plan KTD7). + * + * ANCHOR ALIGNMENT (verified against the scope-extractor): the CALLS `atRange` + * is `nodeToCapture('@reference.call.*', node).range` where the `@reference.call` + * anchor is the WHOLE call/new EXPRESSION node (the callee identifier / member + * property is the `@reference.name` SUB-tag, excluded from the anchor by + * `anchorCaptureFor` + `KNOWN_SUB_TAGS`; `atRange: anchor.range` at + * scope-extractor.ts:1030). The harvester's `visitCall`/`visitNew` receives that + * exact `call_expression`/`new_expression` node and records its `startPosition`, + * so `at` == the CALLS atRange for every call shape: + * - bare call `foo(x)` → `at` == the bare callee/call-expr start. + * - member call `arr.map(x)` → `at` == the call-expr start, i.e. the receiver + * `arr`'s position (where `@reference.call.member` + * anchors), NOT the `.map` property token. + * - chained call `a.b.c(x)` → `at` == the outer call-expr start (`a`). + * + * Member-read sites carry no `at` (the resolved-id join only consumes call/new). + */ + +/** The call/new sites of `cfg`, in (block, statement, site) order. */ +function callSites(cfg: FunctionCfg): SiteRecord[] { + return allSites(cfg).filter((s) => s.kind === 'call' || s.kind === 'new'); +} + +/** The single call/new site whose dotted callee path is `callee` (throws otherwise). */ +function siteByCallee(cfg: FunctionCfg, callee: string): SiteRecord { + const matches = callSites(cfg).filter((s) => s.callee === callee); + if (matches.length !== 1) { + throw new Error(`expected exactly 1 call site with callee ${callee}, got ${matches.length}`); + } + return matches[0]; +} + +describe('SiteRecord.at — call-site anchor position (U1)', () => { + it('a bare call `foo(x)` carries `at` of the callee/call-expr anchor', () => { + // `function f(x) { foo(x); }` — `foo(` starts at column 16 on line 1; for a + // bare call the call-expression start and the callee identifier coincide, so + // this is the anchor the CALLS `@reference.call.free` resolution keys on. + const cfg = cfgOf(`function f(x) { foo(x); }`); + expect(siteByCallee(cfg, 'foo')).toMatchObject({ kind: 'call', at: [1, 16] }); + }); + + it('a member call `arr.map(...)` carries `at` of the resolved-reference anchor (the call-expr / receiver start)', () => { + // `function f(arr) { arr.map(x => x); }` — `arr.map(...)` is one + // `call_expression` starting at `arr` (column 18). The CALLS + // `@reference.call.member` anchor is that whole call_expression, so `at` + // records the receiver/call-expr start, NOT the `.map` property token (which + // is only the `@reference.name` sub-tag — plan KTD7). + const cfg = cfgOf(`function f(arr) { arr.map(x => x); }`); + expect(siteByCallee(cfg, 'arr.map')).toMatchObject({ kind: 'call', at: [1, 18] }); + }); + + it('a chained call `a.b.c(x)` carries `at` of the outer call-expr start (the chain root)', () => { + // `a.b.c(x)` is one call_expression starting at the chain root `a` (column + // 19); the `@reference.call.member` anchor and the harvested `at` both land + // there — the outer call-expr start, not the `.c` property token. + const cfg = cfgOf(`function f(a, x) { a.b.c(x); }`); + expect(siteByCallee(cfg, 'a.b.c')).toMatchObject({ kind: 'call', at: [1, 19] }); + }); + + it('an argument-position call on a LATER line carries that call`s own line, not the statement head line', () => { + // line 2: outer( + // line 3: inner(a) + // line 4: ); + // The whole call statement begins on line 2, but the inner argument-position + // call begins on line 3 — its `SiteRecord.at` line MUST be 3 (the inner + // call's line). This is the core bug the position field fixes: a + // statement-line join would mis-attribute `inner`'s resolved id to line 2. + const cfg = cfgOf(`function f(a, b) {\n outer(\n inner(a)\n );\n}`); + const outer = siteByCallee(cfg, 'outer'); + const inner = siteByCallee(cfg, 'inner'); + expect(outer).toMatchObject({ kind: 'call', at: [2, 2] }); + expect(inner).toMatchObject({ kind: 'call', at: [3, 4] }); + // The inner call's `at` line is the INNER call's line, distinct from the + // statement head — assert the inequality unconditionally. + expect(inner.at?.[0]).toBe(3); + expect(inner.at?.[0]).not.toBe(outer.at?.[0]); + }); + + it('a `new` site carries `at` of the new-expression anchor', () => { + // `new User(x)` — `new` starts at column 16; the harvester records the + // new_expression start, matching `@reference.call.constructor`'s anchor. + const cfg = cfgOf(`function f(x) { const u = new User(x); return u; }`); + expect(siteByCallee(cfg, 'User')).toMatchObject({ kind: 'new', at: [1, 26] }); + }); + + it('a nested closure `arr.map(x => foo(x))` records ONLY the outer `arr.map` site — `foo` is excluded', () => { + // The harvester does not record sites inside nested functions (types.ts:150), + // so `foo(x)` inside the arrow produces NO top-level SiteRecord here. The + // block's only call site is the outer `arr.map`, carrying its own `at`. + const cfg = cfgOf(`function f(arr) { arr.map(x => foo(x)); }`); + const sites = callSites(cfg); + // Exactly one call site, and it is the outer member call — not `foo`. + expect(sites).toHaveLength(1); + expect(sites.map((s) => s.callee)).toEqual(['arr.map']); + expect(siteByCallee(cfg, 'arr.map')).toMatchObject({ kind: 'call', at: [1, 18] }); + // `foo` produces no site at this level (nested-fn exclusion preserved). + expect(sites.some((s) => s.callee === 'foo')).toBe(false); + }); + + it('member-read sites carry no `at` (the id join only consumes call/new)', () => { + // `sink(req.body)` — `req.body` is a value-position member-read site; only + // the `sink(...)` call site needs an anchor for the resolved-id join. + const cfg = cfgOf(`function f(req) { sink(req.body); }`); + const memberReads = allSites(cfg).filter((s) => s.kind === 'member-read'); + expect(memberReads.length).toBeGreaterThan(0); + // No member-read site carries `at` (omit-when-absent for non-call sites). + expect(memberReads.every((s) => s.at === undefined)).toBe(true); + // The enclosing call site does carry one. + expect(siteByCallee(cfg, 'sink')).toMatchObject({ kind: 'call', at: [1, 18] }); + }); +}); diff --git a/gitnexus/test/unit/cfg/anchor-alignment.test.ts b/gitnexus/test/unit/cfg/anchor-alignment.test.ts new file mode 100644 index 000000000..795091782 --- /dev/null +++ b/gitnexus/test/unit/cfg/anchor-alignment.test.ts @@ -0,0 +1,293 @@ +/** + * U6 (#2227 tri-review-2, R4) — per-language anchor-alignment guard. + * + * The resolved-callee-id join (`BasicBlock.calleeIds`) relies on ONE invariant: + * the position the CFG harvester stamps on each call/new site (`SiteRecord.at = + * [1-based line, 0-based col]`) must equal the position the live scope query + * stamps on the matching `@reference.call.*` capture (`ReferenceSite.atRange = + * { startLine (1-based), startCol (0-based) }`). The Phase-4 emit joins resolved + * callee ids onto blocks by EXACT position (U3), so if a scope query's anchor + * node ever drifts for a language, `calleeIds` silently empties for that + * language — and today nothing fails: the alignment is asserted only by the + * hardcoded `at` literals in `harvest.test.ts` and the commit-message claims. + * + * This standing test drives BOTH sides on the SAME source, in-process, per + * language, and asserts byte-equality. It does NOT re-implement the private + * broadest-span `anchorCaptureFor` (KTD3): the exported `extractParsedFile` + * already returns the computed `atRange`, and the exported `makeCfgHarness` + + * `allSites` already return the harvested `site.at`. + * + * Join direction that matters: every harvested call/new site that carries an + * `at` (i.e. is id-joinable) must have a corresponding real `@reference.call` + * anchor in the scope set for that file. The scope side may legitimately + * produce extra refs the harvester doesn't (or vice-versa for member-reads with + * no `at`); we only assert the harvest→scope direction over sites with an `at`. + * + * Coverage: the 6 newly-harvested languages (Python, Dart, Kotlin, Ruby, Rust, + * Swift), each with ≥3 representative call shapes INCLUDING its language-specific + * risk case — most notably Dart, whose member-call anchor is the METHOD-NAME + * node (unique among the 12), and Rust's struct-literal constructor (U4). + */ +import { describe, it, expect } from 'vitest'; +import { createRequire } from 'node:module'; + +import { makeCfgHarness, allSites } from '../../helpers/cfg-harness.js'; +import { requireVendoredGrammar } from '../../../src/core/tree-sitter/vendored-grammars.js'; +import { extractParsedFile } from '../../../src/core/ingestion/scope-extractor-bridge.js'; +import type { CfgVisitor, SiteRecord } from '../../../src/core/ingestion/cfg/types.js'; +import type { SyntaxNode } from '../../../src/core/ingestion/utils/ast-helpers.js'; +import type { LanguageProvider } from '../../../src/core/ingestion/language-provider.js'; + +import { createPythonCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/python.js'; +import { createDartCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/dart.js'; +import { createKotlinCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/kotlin.js'; +import { createRubyCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/ruby.js'; +import { createRustCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/rust.js'; +import { createSwiftCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/swift.js'; + +import { pythonProvider } from '../../../src/core/ingestion/languages/python.js'; +import { dartProvider } from '../../../src/core/ingestion/languages/dart.js'; +import { kotlinProvider } from '../../../src/core/ingestion/languages/kotlin.js'; +import { rubyProvider } from '../../../src/core/ingestion/languages/ruby.js'; +import { rustProvider } from '../../../src/core/ingestion/languages/rust.js'; +import { swiftProvider } from '../../../src/core/ingestion/languages/swift.js'; + +const require = createRequire(import.meta.url); + +/** A tree-sitter grammar, typed as the harness expects it. */ +type Grammar = Parameters[0]; + +/** A `[1-based line, 0-based col]` anchor — the shared base of both sides. */ +type Anchor = readonly [number, number]; + +/** One harvested call/new site that carries an `at` (id-joinable). */ +interface HarvestSite { + readonly at: Anchor; + readonly kind: 'call' | 'new'; + readonly callee: string | undefined; +} + +/** One scope-query call reference (kind 'call', incl. callForm constructor). */ +interface ScopeRef { + readonly at: Anchor; + readonly name: string; + readonly callForm: string | undefined; +} + +/** + * Collect every harvested call/new site that carries an `at`, across every + * function CFG of `src`. Member-reads and any call/new without an `at` (e.g. a + * call rooted on a call result that the resolver cannot id-join) are excluded — + * they are not joinable, so they are out of this contract's scope. + */ +function harvestSites( + grammar: Grammar, + visitor: CfgVisitor, + filePath: string, + src: string, +): HarvestSite[] { + const harness = makeCfgHarness(grammar, visitor, filePath); + const out: HarvestSite[] = []; + for (const cfg of harness.cfgsOf(src)) { + for (const s of allSites(cfg)) { + const at = joinableAt(s); + if (at !== undefined) + out.push({ at, kind: s.kind === 'new' ? 'new' : 'call', callee: s.callee }); + } + } + return out; +} + +/** The `at` anchor of `s` iff it is an id-joinable call/new site, else undefined. */ +function joinableAt(s: SiteRecord): Anchor | undefined { + const isCallOrNew = s.kind === 'call' || s.kind === 'new'; + const at = s.at; + return isCallOrNew && at !== undefined ? [at[0], at[1]] : undefined; +} + +/** + * Run the real scope extractor and collect every call-kind reference site's + * anchor. A Rust struct literal is `kind: 'call'` + `callForm: 'constructor'`, + * so the `kind === 'call'` filter already includes it; we keep `callForm` on the + * record purely for diagnostics. + */ +function scopeCallRefs(provider: LanguageProvider, filePath: string, src: string): ScopeRef[] { + const pf = extractParsedFile(provider, src, filePath); + const sites = pf?.referenceSites ?? []; + return sites + .filter((r) => r.kind === 'call') + .map((r) => ({ + at: [r.atRange.startLine, r.atRange.startCol] as const, + name: r.name, + callForm: r.callForm, + })); +} + +const key = (a: Anchor): string => `${a[0]},${a[1]}`; + +interface LangCase { + readonly name: string; + readonly grammar: Grammar; + readonly visitor: CfgVisitor; + readonly provider: LanguageProvider; + readonly filePath: string; + /** Source with ≥3 call shapes, the last being the language-specific risk case. */ + readonly src: string; + /** Minimum number of joinable harvested sites the fixture is expected to yield. */ + readonly minSites: number; +} + +// Each fixture's final shape is the language-specific anchor risk: +// Python — chained `a.b.c()` (anchor still on the call node, not the leaf). +// Dart — cascade `a..m()` + member `a.m()`: the anchor is the METHOD-NAME +// node (unique among the 12 languages). The whole point of U6. +// Kotlin — safe-call `a?.b()` (anchor on the whole call_expression). +// Ruby — paren-less command `puts x` (anchor on the whole `call` node). +// Rust — `::` path `Foo::bar(x)` AND struct-literal `Point {}` (U4, +// `callForm: 'constructor'`). +// Swift — trailing closure `xs.map { }` (anchor on the call_expression). +const LANGS: readonly LangCase[] = [ + { + name: 'python', + grammar: require('tree-sitter-python') as Grammar, + visitor: createPythonCfgVisitor(), + provider: pythonProvider, + filePath: 'fixture.py', + src: `def f(a, b):\n foo(a)\n obj.m(b)\n a.b.c()\n`, + minSites: 3, + }, + { + name: 'dart', + grammar: requireVendoredGrammar('tree-sitter-dart') as Grammar, + visitor: createDartCfgVisitor(), + provider: dartProvider, + filePath: 'fixture.dart', + src: `void f(a, b) {\n foo(a);\n a.m(b);\n a..n();\n}\n`, + minSites: 3, + }, + { + name: 'kotlin', + grammar: requireVendoredGrammar('tree-sitter-kotlin') as Grammar, + visitor: createKotlinCfgVisitor(), + provider: kotlinProvider, + filePath: 'fixture.kt', + src: `fun f(a: Foo, b: Int) {\n foo(b)\n a.m(b)\n a?.n(b)\n}\n`, + minSites: 3, + }, + { + name: 'ruby', + grammar: require('tree-sitter-ruby') as Grammar, + visitor: createRubyCfgVisitor(), + provider: rubyProvider, + filePath: 'fixture.rb', + src: `def f(a, x)\n foo(x)\n a.m(x)\n puts x\nend\n`, + minSites: 3, + }, + { + name: 'rust', + grammar: require('tree-sitter-rust') as Grammar, + visitor: createRustCfgVisitor(), + provider: rustProvider, + filePath: 'fixture.rs', + src: `fn f(a: A, x: i32) {\n foo(x);\n a.m(x);\n Foo::bar(x);\n let p = Point { x: 1 };\n}\n`, + minSites: 4, + }, + { + name: 'swift', + grammar: requireVendoredGrammar('tree-sitter-swift') as Grammar, + visitor: createSwiftCfgVisitor(), + provider: swiftProvider, + filePath: 'fixture.swift', + src: `func f(a: Foo, xs: [Int]) {\n foo(a)\n a.m()\n xs.map { x in x }\n}\n`, + minSites: 3, + }, +]; + +describe('CFG harvester ↔ scope-query anchor alignment (U6/R4)', () => { + for (const lang of LANGS) { + describe(lang.name, () => { + const harvest = harvestSites(lang.grammar, lang.visitor, lang.filePath, lang.src); + const scope = scopeCallRefs(lang.provider, lang.filePath, lang.src); + const scopeKeys = new Set(scope.map((r) => key(r.at))); + + it('the fixture yields the expected number of joinable harvested sites', () => { + // A guard against a vacuous pass: if the harvester silently stopped + // producing sites for this language, `arrayContaining([])` below would + // trivially hold. Pin the floor so the alignment assertion has teeth. + expect(harvest.length).toBeGreaterThanOrEqual(lang.minSites); + }); + + it('every harvested call/new anchor equals a real @reference.call anchor', () => { + // The id-join lands iff each harvested `site.at` is also a scope-query + // call anchor. Compare the harvest anchors against the scope anchor set: + // a drift in either anchor node fails loudly with the offending position. + const harvestKeys = harvest.map((h) => key(h.at)).sort(); + const presentInScope = harvest + .filter((h) => scopeKeys.has(key(h.at))) + .map((h) => key(h.at)) + .sort(); + // `toEqual` (not `arrayContaining`) so an extra harvested anchor that the + // scope query lacks fails — both lists must be identical. + expect(presentInScope).toEqual(harvestKeys); + }); + + it('reports each harvested site with its matched scope ref (diagnostic lock)', () => { + // A second, shape-explicit assertion: each joinable harvest site maps to + // a scope ref AT THE SAME ANCHOR. Builds the joined pairs unconditionally + // (no `if`-guard around an expect) and asserts the full set, so a + // mismatch surfaces the language, the callee, and both positions. + const matched = harvest.map((h) => { + const ref = scope.find((r) => key(r.at) === key(h.at)); + return { + harvestAt: key(h.at), + kind: h.kind, + callee: h.callee, + scopeAt: ref === undefined ? 'NO-SCOPE-ANCHOR' : key(ref.at), + }; + }); + // Every entry's scopeAt must equal its harvestAt — none may be the + // NO-SCOPE-ANCHOR sentinel. + expect(matched.map((m) => m.scopeAt)).toEqual(matched.map((m) => m.harvestAt)); + }); + }); + } + + it('Dart member + cascade calls anchor on the METHOD-NAME node (the unique case)', () => { + // Dart is the ONLY language whose member-call anchor is the method-name + // identifier rather than the whole call / receiver. This pins that the + // harvester and scope query agree on the method-name column specifically + // (col 4 for `a.m`, col 5 for `a..n`), not merely "some shared position". + const dart = LANGS.find((l) => l.name === 'dart')!; + const harvest = harvestSites(dart.grammar, dart.visitor, dart.filePath, dart.src); + const scope = scopeCallRefs(dart.provider, dart.filePath, dart.src); + // member call `a.m(b)` on line 3 anchors on the method name `m` (col 4). + const memberHarvest = harvest.find((h) => h.at[0] === 3)!; + const memberScope = scope.find((r) => r.at[0] === 3)!; + expect(memberHarvest.at).toEqual([3, 4]); + expect([memberScope.at[0], memberScope.at[1]]).toEqual([3, 4]); + expect(memberScope.name).toBe('m'); + // cascade call `a..n()` on line 4 anchors on the method name `n` (col 5). + const cascadeHarvest = harvest.find((h) => h.at[0] === 4)!; + const cascadeScope = scope.find((r) => r.at[0] === 4)!; + expect(cascadeHarvest.at).toEqual([4, 5]); + expect([cascadeScope.at[0], cascadeScope.at[1]]).toEqual([4, 5]); + expect(cascadeScope.name).toBe('n'); + }); + + it('Rust struct-literal constructor anchor aligns (U4 → @reference.call.constructor)', () => { + // The struct literal `Point { x: 1 }` is a `kind: 'new'` harvest site; the + // scope query tags it `kind: 'call'` + `callForm: 'constructor'`. Both must + // anchor on the struct_expression start so the resolved constructor id joins. + const rust = LANGS.find((l) => l.name === 'rust')!; + const harvest = harvestSites(rust.grammar, rust.visitor, rust.filePath, rust.src); + const scope = scopeCallRefs(rust.provider, rust.filePath, rust.src); + const structHarvest = harvest.find((h) => h.kind === 'new')!; + const structScope = scope.find((r) => r.callForm === 'constructor')!; + expect(structHarvest.callee).toBe('Point'); + expect([structScope.at[0], structScope.at[1]]).toEqual([ + structHarvest.at[0], + structHarvest.at[1], + ]); + expect(structScope.name).toBe('Point'); + }); +}); diff --git a/gitnexus/test/unit/cfg/harvest.test.ts b/gitnexus/test/unit/cfg/harvest.test.ts index a77a36149..6d42ea206 100644 --- a/gitnexus/test/unit/cfg/harvest.test.ts +++ b/gitnexus/test/unit/cfg/harvest.test.ts @@ -778,3 +778,1085 @@ describe('U11 — per-statement site cap (defensive bound on harvested sites[])' expect(flat.some((e) => Array.isArray(e) && e[1] === -1)).toBe(false); }); }); + +// ── #2227 follow-up — Python call-site harvest (pilot language) ────────────── +// Drives the REAL Python CFG visitor (PythonHarvester) against real source via +// the language-agnostic harness, mirroring the TS site tests above. The `at` +// anchor is the `call` node's start position (byte-aligned with the +// `@reference.call.*` CALLS anchor so the resolved-id join lands). + +import { createRequire } from 'node:module'; +import { makeCfgHarness, type CfgHarness } from '../../helpers/cfg-harness.js'; +import { createPythonCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/python.js'; + +const pyGrammar = createRequire(import.meta.url)('tree-sitter-python') as Parameters< + typeof makeCfgHarness +>[0]; +const py: CfgHarness = makeCfgHarness(pyGrammar, createPythonCfgVisitor(), 'fixture.py'); + +describe('Python call-site harvest', () => { + it('foo(a, b) → one call site, positions 0→[a], 1→[b], at the call node line/col', () => { + const cfg = py.cfgOf(`def f(a, b):\n foo(a, b)\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('foo'); + expect(s.receiver).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + expect(s.parent).toBeUndefined(); + // `at` is the `call` node start: line 2 (1-based), col 4 (after indent). + expect(s.at).toEqual([2, 4]); + }); + + it('obj.method(x) → dotted callee path + receiver = obj', () => { + const cfg = py.cfgOf(`def f(obj, x):\n obj.method(x)\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('obj.method'); + expect(s.receiver).toBe(bindingIdx(cfg, 'obj')); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + // chain-length-1 callee: the access IS the callee — no member-read site + expect(siteFact(cfg, 2).sites).toHaveLength(1); + // and the receiver use is recorded exactly once (no double-record) + expect(siteFact(cfg, 2).uses.filter((u) => u === bindingIdx(cfg, 'obj'))).toHaveLength(1); + // `at` is the call node start (the receiver `obj`), col 4. + expect(s.at).toEqual([2, 4]); + }); + + it('a.b.c() → callee path a.b.c, receiver = root a, plus a mid-chain member read', () => { + const cfg = py.cfgOf(`def f(a):\n a.b.c()\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.callee).toBe('a.b.c'); + expect(call.receiver).toBe(bindingIdx(cfg, 'a')); + // the innermost access `a.b` (the non-callee load) is a member read + const read = sites.find((s) => s.kind === 'member-read')!; + expect(read.object).toBe(bindingIdx(cfg, 'a')); + expect(read.property).toBe('b'); + }); + + it("multi-line call → the site's `at` line is the call node's start line", () => { + const cfg = py.cfgOf(`def f(a, b):\n foo(\n a,\n b,\n )\n`); + const s = siteFact(cfg).sites![0]; + expect(s.callee).toBe('foo'); + // `call` starts on line 2 even though args span lines 3-4. + expect(s.at![0]).toBe(2); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + }); + + it('exec(*args) → spread index recorded, args binding occurs at the position', () => { + const cfg = py.cfgOf(`def f(*args):\n exec(*args)\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('exec'); + expect(s.spread).toBe(0); + expect(s.args).toEqual([[bindingIdx(cfg, 'args')]]); + }); + + it('two calls in one function → two distinct call sites', () => { + const cfg = py.cfgOf(`def f(a, b):\n foo(a)\n bar(b)\n`); + const calls = allSites(cfg).filter((s) => s.kind === 'call'); + expect(calls.map((s) => s.callee).sort()).toEqual(['bar', 'foo']); + }); + + it('x = f(y) → resultDefs carries x; nested escape(req.body) tags + member read', () => { + const cfg = py.cfgOf(`def f(y):\n x = g(y)\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('g'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'y')]]); + }); + + it('nested call exec(escape(x)) → inner site parent-linked, occurrence via-tagged', () => { + const cfg = py.cfgOf(`def f(x):\n exec(escape(x))\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(2); + const execIdx = sites.findIndex((s) => s.callee === 'exec'); + const escapeIdx = sites.findIndex((s) => s.callee === 'escape'); + const x = bindingIdx(cfg, 'x'); + expect(sites[escapeIdx].args).toEqual([[x]]); + expect(sites[escapeIdx].parent).toEqual([execIdx, 0]); + expect(sites[execIdx].args).toEqual([[[x, escapeIdx]]]); + }); + + it('keyword argument f(k=v) → only the value v is an occurrence (key is not a use)', () => { + const cfg = py.cfgOf(`def f(v):\n foo(k=v)\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('foo'); + // `k` mints no binding; `v` occurs at position 0. + expect(bindingIdxs(cfg, 'k')).toHaveLength(0); + expect(s.args).toEqual([[bindingIdx(cfg, 'v')]]); + }); + + it('def/use facts stay intact alongside the new sites (regression guard)', () => { + const cfg = py.cfgOf(`def f(y):\n x = g(y)\n use(x)\n`); + const x = bindingIdx(cfg, 'x'); + const y = bindingIdx(cfg, 'y'); + expect(defsOf(cfg)).toContain(x); + expect(usesOf(cfg)).toContain(y); + expect(usesOf(cfg)).toContain(x); + }); +}); + +// ── Dart call-site harvest ────────────────────────────────────────────────── +// Drives the REAL Dart CFG visitor (DartHarvester) against real source via the +// language-agnostic harness, mirroring the Python site tests above. Dart has NO +// `call_expression` node — a call is a FLAT SIBLING RUN (`identifier` + +// `selector*`); a `selector(argument_part)` is the call marker. The `at` anchor +// is byte-aligned with the Dart `@reference.call.*` CALLS anchor, which the +// scope-extractor places on the callee NAME identifier: the callee identifier +// for a free / implicit-constructor call (`foo`/`Foo`), and the METHOD-name +// identifier for a member call (`.method`'s `method`, NOT the receiver). The +// grammar is VENDORED (loaded from vendor/ like the Kotlin/Swift tests). + +import { requireVendoredGrammar } from '../../../src/core/tree-sitter/vendored-grammars.js'; +import { createDartCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/dart.js'; + +const dartGrammar = requireVendoredGrammar('tree-sitter-dart') as Parameters< + typeof makeCfgHarness +>[0]; +const dart: CfgHarness = makeCfgHarness(dartGrammar, createDartCfgVisitor(), 'fixture.dart'); + +describe('Dart call-site harvest', () => { + it('foo(a, b) → one call site, positions 0→[a], 1→[b], at the callee id line/col', () => { + const cfg = dart.cfgOf(`void f(a, b) {\n foo(a, b);\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('foo'); + expect(s.receiver).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + expect(s.parent).toBeUndefined(); + // `at` is the callee identifier `foo`: line 2 (1-based), col 2 (after indent). + expect(s.at).toEqual([2, 2]); + }); + + it('obj.method(x) → dotted callee path + receiver = obj, at the METHOD name', () => { + const cfg = dart.cfgOf(`void f(obj, x) {\n obj.method(x);\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.callee).toBe('obj.method'); + expect(s.receiver).toBe(bindingIdx(cfg, 'obj')); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + // chain-length-1 callee: the access IS the callee — no member-read site. + expect(siteFact(cfg, 2).uses.filter((u) => u === bindingIdx(cfg, 'obj'))).toHaveLength(1); + // `at` is the method-name identifier `method` (col 6), NOT the receiver `obj`. + expect(s.at).toEqual([2, 6]); + }); + + it('a.b.c() → callee path a.b.c, receiver = root a, plus a mid-chain member read', () => { + const cfg = dart.cfgOf(`void f(a) {\n a.b.c();\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.callee).toBe('a.b.c'); + expect(call.receiver).toBe(bindingIdx(cfg, 'a')); + // `at` is the call-name identifier `c` (col 6). + expect(call.at).toEqual([2, 6]); + // the innermost access `a.b` (the non-callee load) is a member read. + const read = sites.find((s) => s.kind === 'member-read')!; + expect(read.object).toBe(bindingIdx(cfg, 'a')); + expect(read.property).toBe('b'); + }); + + it("multi-line call → the site's `at` line is the callee identifier's line", () => { + const cfg = dart.cfgOf(`void f(a, b) {\n foo(\n a,\n b,\n );\n}\n`); + const s = siteFact(cfg).sites![0]; + expect(s.callee).toBe('foo'); + // `foo` is on line 2 even though the args span lines 3-4. + expect(s.at![0]).toBe(2); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + }); + + it('two calls in one function → two distinct call sites', () => { + const cfg = dart.cfgOf(`void f(a, b) {\n foo(a);\n bar(b);\n}\n`); + const calls = allSites(cfg).filter((s) => s.kind === 'call'); + expect(calls.map((s) => s.callee).sort()).toEqual(['bar', 'foo']); + }); + + it('var x = g(y) → resultDefs carries x; arg y occurs at position 0', () => { + const cfg = dart.cfgOf(`void f(y) {\n var x = g(y);\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('g'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'y')]]); + // `at` is the callee identifier `g` (col 10, after ` var x = `). + expect(s.at).toEqual([2, 10]); + }); + + it('nested call exec(escape(x)) → inner site parent-linked, occurrence via-tagged', () => { + const cfg = dart.cfgOf(`void f(x) {\n exec(escape(x));\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(2); + const execIdx = sites.findIndex((s) => s.callee === 'exec'); + const escapeIdx = sites.findIndex((s) => s.callee === 'escape'); + const x = bindingIdx(cfg, 'x'); + expect(sites[escapeIdx].args).toEqual([[x]]); + expect(sites[escapeIdx].parent).toEqual([execIdx, 0]); + expect(sites[execIdx].args).toEqual([[[x, escapeIdx]]]); + }); + + it('implicit constructor Foo(1) → kind call (Dart-2 implicit, structurally a free call)', () => { + const cfg = dart.cfgOf(`void f() {\n var x = Foo(1);\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('Foo'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + // anchored on the callee identifier `Foo` (col 10) — matches + // `@reference.call.constructor`, which the resolution keys on the same node. + expect(s.at).toEqual([2, 10]); + }); + + it('new Foo(1) → kind new (the only single-node call shape Dart has)', () => { + const cfg = dart.cfgOf(`void f() {\n var x = new Foo(1);\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.kind).toBe('new'); + expect(s.callee).toBe('Foo'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + // the type identifier `Foo` is a type, not a scalar binding — no use of it. + expect(bindingIdxs(cfg, 'Foo')).toHaveLength(0); + }); + + it('named argument foo(k: v) → only the value v is an occurrence (key is not a use)', () => { + const cfg = dart.cfgOf(`void f(v) {\n foo(k: v);\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('foo'); + // `k` mints no binding; `v` occurs at position 0. + expect(bindingIdxs(cfg, 'k')).toHaveLength(0); + expect(s.args).toEqual([[bindingIdx(cfg, 'v')]]); + }); + + it('def/use facts stay intact alongside the new sites (regression guard)', () => { + const cfg = dart.cfgOf(`void f(y) {\n var x = g(y);\n use(x);\n}\n`); + const x = bindingIdx(cfg, 'x'); + const y = bindingIdx(cfg, 'y'); + // the binding table + def/use are coherent: x is defined, y and x are used. + expect(defsOf(cfg)).toContain(x); + expect(usesOf(cfg)).toContain(y); + expect(usesOf(cfg)).toContain(x); + expect(cfg.bindings![y].kind).toBe('param'); + }); +}); + +// ── Kotlin call-site harvest ───────────────────────────────────────────────── +// Drives the REAL Kotlin CFG visitor (KotlinHarvester) against real source via +// the language-agnostic harness, mirroring the Python / Dart site tests above. +// A Kotlin call is a `call_expression` whose last child is a `call_suffix` +// (holding `value_arguments` / a trailing `annotated_lambda`); the callee is the +// preceding `simple_identifier` (free) or `navigation_expression` (member / +// chained / safe-call). Kotlin has no `new` — constructor calls are ordinary +// `call_expression`s (`kind: 'call'`). The `at` anchor is byte-aligned with the +// Kotlin `@reference.call.free/.member` CALLS anchor, which the scope query +// places on the WHOLE `call_expression` node (the Go/Python whole-call model, +// NOT Dart's callee-name model) — so for a member/chained call `at` starts at the +// RECEIVER, exactly where the CALLS anchor starts. The grammar is VENDORED +// (loaded from vendor/ like the Dart/Swift tests). + +import { createKotlinCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/kotlin.js'; + +const kotlinGrammar = requireVendoredGrammar('tree-sitter-kotlin') as Parameters< + typeof makeCfgHarness +>[0]; +const kotlin: CfgHarness = makeCfgHarness(kotlinGrammar, createKotlinCfgVisitor(), 'fixture.kt'); + +describe('Kotlin call-site harvest', () => { + it('foo(a, b) → one call site, positions 0→[a], 1→[b], at the call_expression line/col', () => { + const cfg = kotlin.cfgOf(`fun f(a: Int, b: Int) {\n foo(a, b)\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('foo'); + expect(s.receiver).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + expect(s.parent).toBeUndefined(); + // `at` is the `call_expression` node start: line 2 (1-based), col 4 (indent). + expect(s.at).toEqual([2, 4]); + }); + + it('obj.method(x) → dotted callee path + receiver = obj, at the call_expression start', () => { + const cfg = kotlin.cfgOf(`fun f(obj: Foo, x: Int) {\n obj.method(x)\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.callee).toBe('obj.method'); + expect(s.receiver).toBe(bindingIdx(cfg, 'obj')); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + // chain-length-1 callee: the access IS the callee — no member-read site. + expect(siteFact(cfg, 2).uses.filter((u) => u === bindingIdx(cfg, 'obj'))).toHaveLength(1); + // `at` is the call_expression start (the receiver `obj`), col 4 — NOT the + // method name. The Kotlin CALLS anchor keys on the same call_expression node. + expect(s.at).toEqual([2, 4]); + }); + + it('a.b.c() → callee path a.b.c, receiver = root a, plus a mid-chain member read', () => { + const cfg = kotlin.cfgOf(`fun f(a: Foo) {\n a.b.c()\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.callee).toBe('a.b.c'); + expect(call.receiver).toBe(bindingIdx(cfg, 'a')); + // `at` is the whole call_expression start (the root `a`), col 4. + expect(call.at).toEqual([2, 4]); + // the innermost access `a.b` (the non-callee load) is a member read. + const read = sites.find((s) => s.kind === 'member-read')!; + expect(read.object).toBe(bindingIdx(cfg, 'a')); + expect(read.property).toBe('b'); + }); + + it('a?.b() safe-call → still a call site (callee a.b, receiver = a)', () => { + const cfg = kotlin.cfgOf(`fun f(a: Foo?) {\n a?.b()\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.kind).toBe('call'); + expect(call.callee).toBe('a.b'); + expect(call.receiver).toBe(bindingIdx(cfg, 'a')); + expect(call.at).toEqual([2, 4]); + }); + + it("multi-line call → the site's `at` line is the call_expression's start line", () => { + const cfg = kotlin.cfgOf( + `fun f(a: Int, b: Int) {\n foo(\n a,\n b\n )\n}\n`, + ); + const s = siteFact(cfg).sites![0]; + expect(s.callee).toBe('foo'); + // `call_expression` starts on line 2 even though the args span lines 3-4. + expect(s.at![0]).toBe(2); + expect(s.at![1]).toBe(4); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + }); + + it('two calls in one function → two distinct call sites', () => { + const cfg = kotlin.cfgOf(`fun f(a: Int, b: Int) {\n foo(a)\n bar(b)\n}\n`); + const calls = allSites(cfg).filter((s) => s.kind === 'call'); + expect(calls.map((s) => s.callee).sort()).toEqual(['bar', 'foo']); + }); + + it('val x = g(y) → resultDefs carries x; arg y occurs at position 0', () => { + const cfg = kotlin.cfgOf(`fun f(y: Int) {\n val x = g(y)\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('g'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'y')]]); + // `at` is the call_expression start `g` (col 12, after ` val x = `). + expect(s.at).toEqual([2, 12]); + }); + + it('x = g(y) assignment → resultDefs carries x (plain = scalar lvalue)', () => { + const cfg = kotlin.cfgOf(`fun f(y: Int) {\n var x = 0\n x = g(y)\n}\n`); + const s = siteFact(cfg, 3).sites![0]; + expect(s.callee).toBe('g'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'y')]]); + }); + + it('nested call exec(escape(x)) → inner site parent-linked, occurrence via-tagged', () => { + const cfg = kotlin.cfgOf(`fun f(x: Int) {\n exec(escape(x))\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(2); + const execIdx = sites.findIndex((s) => s.callee === 'exec'); + const escapeIdx = sites.findIndex((s) => s.callee === 'escape'); + const x = bindingIdx(cfg, 'x'); + expect(sites[escapeIdx].args).toEqual([[x]]); + expect(sites[escapeIdx].parent).toEqual([execIdx, 0]); + expect(sites[execIdx].args).toEqual([[[x, escapeIdx]]]); + }); + + it('exec(*args) → spread index recorded, args binding occurs at the position', () => { + const cfg = kotlin.cfgOf(`fun f(args: IntArray) {\n exec(*args)\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('exec'); + expect(s.spread).toBe(0); + expect(s.args).toEqual([[bindingIdx(cfg, 'args')]]); + }); + + it('named argument foo(k = v) → only the value v is an occurrence (name is not a use)', () => { + const cfg = kotlin.cfgOf(`fun f(v: Int) {\n foo(k = v)\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('foo'); + // `k` mints no binding; `v` occurs at position 0. + expect(bindingIdxs(cfg, 'k')).toHaveLength(0); + expect(s.args).toEqual([[bindingIdx(cfg, 'v')]]); + }); + + it('req.body value read → a member-read site (no call), object = req', () => { + const cfg = kotlin.cfgOf(`fun f(req: Req) {\n val b = req.body\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const read = sites.find((s) => s.kind === 'member-read')!; + expect(read.object).toBe(bindingIdx(cfg, 'req')); + expect(read.property).toBe('body'); + expect(sites.filter((s) => s.kind === 'call')).toHaveLength(0); + }); + + it('def/use facts stay intact alongside the new sites (regression guard)', () => { + const cfg = kotlin.cfgOf(`fun f(y: Int) {\n val x = g(y)\n use(x)\n}\n`); + const x = bindingIdx(cfg, 'x'); + const y = bindingIdx(cfg, 'y'); + // the binding table + def/use are coherent: x is defined, y and x are used. + expect(defsOf(cfg)).toContain(x); + expect(usesOf(cfg)).toContain(y); + expect(usesOf(cfg)).toContain(x); + expect(cfg.bindings![y].kind).toBe('param'); + }); +}); + +// ── Ruby call-site harvest ─────────────────────────────────────────────────── +// Drives the REAL Ruby CFG visitor (RubyHarvester) against real source via the +// language-agnostic harness, mirroring the Python / Kotlin site tests above. +// EVERY Ruby call is a single `call` node (fields receiver?/method/arguments?): +// a free call `foo(a)`, an implicit-receiver paren-less command `puts x` / +// `attr_accessor :x`, a member call `obj.method(x)`, a safe-call `obj&.m()`, and +// a chained `a.b.c` (nested `call` receivers) are all `call` nodes — there is NO +// `command` node in this grammar. Ruby has no `new` (`Foo.new` is a member call), +// so every site is `kind: 'call'`. A receiver-only no-args `call` (`obj.field`) +// is grammatically a member call and the CALLS query tags it +// `@reference.call.member`, so it is a call site (NOT a member-read). The `at` +// anchor is byte-aligned with the Ruby `@reference.call.free/.member` CALLS +// anchor, which the scope query places on the WHOLE `call` node (the Go/Python/ +// Kotlin whole-call model) — so for a member/chained call `at` starts at the +// RECEIVER. The grammar is loaded via `require` (like Python). + +import { createRubyCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/ruby.js'; + +const rubyGrammar = createRequire(import.meta.url)('tree-sitter-ruby') as Parameters< + typeof makeCfgHarness +>[0]; +const ruby: CfgHarness = makeCfgHarness(rubyGrammar, createRubyCfgVisitor(), 'fixture.rb'); + +describe('Ruby call-site harvest', () => { + it('foo(a, b) → one call site, positions 0→[a], 1→[b], at the call node line/col', () => { + const cfg = ruby.cfgOf(`def f(a, b)\n foo(a, b)\nend\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('foo'); + expect(s.receiver).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + expect(s.parent).toBeUndefined(); + // `at` is the `call` node start: line 2 (1-based), col 2 (after ` ` indent). + expect(s.at).toEqual([2, 2]); + }); + + it('obj.method(x) → dotted callee path + receiver = obj, at the call node start', () => { + const cfg = ruby.cfgOf(`def f(obj, x)\n obj.method(x)\nend\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.callee).toBe('obj.method'); + expect(s.receiver).toBe(bindingIdx(cfg, 'obj')); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + // the receiver use is recorded exactly once (no double-record). + expect(siteFact(cfg, 2).uses.filter((u) => u === bindingIdx(cfg, 'obj'))).toHaveLength(1); + // `at` is the call node start (the receiver `obj`), col 2 — NOT the method + // name. The Ruby CALLS anchor keys on the same whole `call` node. + expect(s.at).toEqual([2, 2]); + }); + + it('a.b.c chained → callee path a.b.c, receiver = root a, plus a mid-chain member read', () => { + const cfg = ruby.cfgOf(`def f(a)\n a.b.c\nend\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.callee).toBe('a.b.c'); + expect(call.receiver).toBe(bindingIdx(cfg, 'a')); + // `at` is the whole `call` node start (the root `a`), col 2. + expect(call.at).toEqual([2, 2]); + // the innermost access `a.b` (the non-callee load) is a member read. + const read = sites.find((s) => s.kind === 'member-read')!; + expect(read.object).toBe(bindingIdx(cfg, 'a')); + expect(read.property).toBe('b'); + }); + + it('implicit-receiver paren-less command puts x → site callee `puts`, arg x at position 0', () => { + const cfg = ruby.cfgOf(`def f(x)\n puts x\nend\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('puts'); + expect(s.receiver).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + expect(s.at).toEqual([2, 2]); + }); + + it('attr_accessor :x command → site callee `attr_accessor` (symbol key is not an occurrence)', () => { + const cfg = ruby.cfgOf(`def f\n attr_accessor :x\nend\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('attr_accessor'); + // the `:x` symbol argument mints no binding and is not a value occurrence. + expect(bindingIdxs(cfg, 'x')).toHaveLength(0); + expect(s.args).toBeUndefined(); + }); + + it('obj&.m() safe-navigation → still a call site (callee obj.m, receiver = obj)', () => { + const cfg = ruby.cfgOf(`def f(obj)\n obj&.m()\nend\n`); + const s = siteFact(cfg, 2).sites!.find((x) => x.kind === 'call')!; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('obj.m'); + expect(s.receiver).toBe(bindingIdx(cfg, 'obj')); + expect(s.at).toEqual([2, 2]); + }); + + it('obj.field (no-arg member) → harvested as a call site (matches @reference.call.member)', () => { + const cfg = ruby.cfgOf(`def f(obj)\n y = obj.field\nend\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.callee).toBe('obj.field'); + expect(call.receiver).toBe(bindingIdx(cfg, 'obj')); + // a single-access receiver is the callee itself — no separate member-read. + expect(sites.filter((s) => s.kind === 'member-read')).toHaveLength(0); + }); + + it("multi-line call → the site's `at` line is the call node's start line", () => { + const cfg = ruby.cfgOf(`def f(a, b)\n foo(\n a,\n b,\n )\nend\n`); + const s = siteFact(cfg).sites![0]; + expect(s.callee).toBe('foo'); + // `call` starts on line 2 even though the args span lines 3-4. + expect(s.at![0]).toBe(2); + expect(s.at![1]).toBe(2); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + }); + + it('two calls in one function → two distinct call sites', () => { + const cfg = ruby.cfgOf(`def f(a, b)\n foo(a)\n bar(b)\nend\n`); + const calls = allSites(cfg).filter((s) => s.kind === 'call'); + expect(calls.map((s) => s.callee).sort()).toEqual(['bar', 'foo']); + }); + + it('x = g(y) → resultDefs carries x; arg y occurs at position 0', () => { + const cfg = ruby.cfgOf(`def f(y)\n x = g(y)\nend\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('g'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'y')]]); + // `at` is the `call` node start `g` (col 6, after ` x = `). + expect(s.at).toEqual([2, 6]); + }); + + it('nested call exec(escape(x)) → inner site parent-linked, occurrence via-tagged', () => { + const cfg = ruby.cfgOf(`def f(x)\n exec(escape(x))\nend\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(2); + const execIdx = sites.findIndex((s) => s.callee === 'exec'); + const escapeIdx = sites.findIndex((s) => s.callee === 'escape'); + const x = bindingIdx(cfg, 'x'); + expect(sites[escapeIdx].args).toEqual([[x]]); + expect(sites[escapeIdx].parent).toEqual([execIdx, 0]); + expect(sites[execIdx].args).toEqual([[[x, escapeIdx]]]); + }); + + it('exec(*xs) → spread index recorded, xs binding occurs at the position', () => { + const cfg = ruby.cfgOf(`def f(xs)\n exec(*xs)\nend\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('exec'); + expect(s.spread).toBe(0); + expect(s.args).toEqual([[bindingIdx(cfg, 'xs')]]); + }); + + it('keyword argument foo(k: v) → only the value v is an occurrence (key is not a use)', () => { + const cfg = ruby.cfgOf(`def f(v)\n foo(k: v)\nend\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('foo'); + // `k` mints no binding; `v` occurs at position 0. + expect(bindingIdxs(cfg, 'k')).toHaveLength(0); + expect(s.args).toEqual([[bindingIdx(cfg, 'v')]]); + }); + + it('a do/{} block argument is opaque — not an arg occurrence (one site for the call)', () => { + const cfg = ruby.cfgOf(`def f(xs)\n xs.map { |i| g(i) }\nend\n`); + // the outer `xs.map` is one call site anchored on line 2; the block body's + // `g(i)` is the block's OWN CFG (opaque), so it is NOT an argument occurrence + // of `xs.map` (filter by the call's `at` line, not the function-relative one). + const onLine2 = allSites(cfg).filter((s) => (s.at?.[0] ?? -1) === 2); + expect(onLine2).toHaveLength(1); + expect(onLine2[0].callee).toBe('xs.map'); + expect(onLine2[0].receiver).toBe(bindingIdx(cfg, 'xs')); + expect(onLine2[0].args).toBeUndefined(); + }); + + it('def/use facts stay intact alongside the new sites (regression guard)', () => { + const cfg = ruby.cfgOf(`def f(y)\n x = g(y)\n use(x)\nend\n`); + const x = bindingIdx(cfg, 'x'); + const y = bindingIdx(cfg, 'y'); + // the binding table + def/use are coherent: x is defined, y and x are used. + expect(defsOf(cfg)).toContain(x); + expect(usesOf(cfg)).toContain(y); + expect(usesOf(cfg)).toContain(x); + expect(cfg.bindings![y].kind).toBe('param'); + }); + + it('block param def/use are unchanged by the site harvest (block-handling regression)', () => { + const blk = ruby.cfgsOf(`def f(xs)\n xs.map { |item| item * 2 }\nend\n`)[1]; + const item = bindingIdx(blk, 'item'); + expect(defsOf(blk)).toContain(item); + expect(usesOf(blk)).toContain(item); + }); +}); + +// ── Swift call-site harvest ────────────────────────────────────────────────── +// Drives the REAL Swift CFG visitor (SwiftHarvester) against real source via the +// language-agnostic harness, mirroring the Kotlin / Dart site tests above. A +// Swift call is a `call_expression` whose last child is a `call_suffix` (holding +// `value_arguments` / a trailing closure `lambda_literal`); the callee is the +// preceding `simple_identifier` (free / init) or `navigation_expression` (member +// / chained / optional-chain). Swift has no `new` — an init call `Foo(...)` is an +// ordinary `call_expression` (`kind: 'call'`). The `at` anchor is byte-aligned +// with the Swift `@reference.call.free/.member/.constructor` CALLS anchor, which +// the scope query places on the WHOLE `call_expression` node (the Kotlin/Go/ +// Python whole-call model, NOT Dart's callee-name model) — so for a member / +// chained call `at` starts at the RECEIVER, exactly where the CALLS anchor +// starts. The grammar is VENDORED (loaded from vendor/ like the Dart/Kotlin +// tests). Each `at` below was confirmed byte-equal to the real +// `getSwiftScopeQuery` `@reference.call.*` atRange. + +import { createSwiftCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/swift.js'; + +const swiftGrammar = requireVendoredGrammar('tree-sitter-swift') as Parameters< + typeof makeCfgHarness +>[0]; +const swift: CfgHarness = makeCfgHarness(swiftGrammar, createSwiftCfgVisitor(), 'fixture.swift'); + +describe('Swift call-site harvest', () => { + it('foo(a, b) → one call site, positions 0→[a], 1→[b], at the call_expression line/col', () => { + const cfg = swift.cfgOf(`func f(a: Int, b: Int) {\n foo(a, b)\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('foo'); + expect(s.receiver).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + expect(s.parent).toBeUndefined(); + // `at` is the `call_expression` node start: line 2 (1-based), col 4 (indent). + expect(s.at).toEqual([2, 4]); + }); + + it('obj.method(x) → dotted callee path + receiver = obj, at the call_expression start', () => { + const cfg = swift.cfgOf(`func f(obj: Foo, x: Int) {\n obj.method(x)\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.callee).toBe('obj.method'); + expect(s.receiver).toBe(bindingIdx(cfg, 'obj')); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + // chain-length-1 callee: the access IS the callee — no member-read site, and + // the receiver use is recorded exactly once. + expect(siteFact(cfg, 2).uses.filter((u) => u === bindingIdx(cfg, 'obj'))).toHaveLength(1); + // `at` is the call_expression start (the receiver `obj`), col 4 — NOT the + // method name. The Swift CALLS anchor keys on the same call_expression node. + expect(s.at).toEqual([2, 4]); + }); + + it('a.b.c() → callee path a.b.c, receiver = root a, plus a mid-chain member read', () => { + const cfg = swift.cfgOf(`func f(a: Foo) {\n a.b.c()\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.callee).toBe('a.b.c'); + expect(call.receiver).toBe(bindingIdx(cfg, 'a')); + // `at` is the whole call_expression start (the root `a`), col 4. + expect(call.at).toEqual([2, 4]); + // the innermost access `a.b` (the non-callee load) is a member read. + const read = sites.find((s) => s.kind === 'member-read')!; + expect(read.object).toBe(bindingIdx(cfg, 'a')); + expect(read.property).toBe('b'); + }); + + it('a?.b() optional-chain call → still a call site (callee a.b, receiver = a)', () => { + const cfg = swift.cfgOf(`func f(a: Foo?) {\n a?.b()\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.kind).toBe('call'); + expect(call.callee).toBe('a.b'); + expect(call.receiver).toBe(bindingIdx(cfg, 'a')); + expect(call.at).toEqual([2, 4]); + }); + + it('trailing closure xs.map { … } → one site for xs.map; the closure body is not an arg', () => { + const cfg = swift.cfgOf(`func f(xs: [Int]) {\n xs.map { x in x + 1 }\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.callee).toBe('xs.map'); + expect(call.receiver).toBe(bindingIdx(cfg, 'xs')); + // The trailing closure is a nested function body (opaque) — NOT an argument + // occurrence, so the site records no args. + expect(call.args).toBeUndefined(); + expect(call.at).toEqual([2, 4]); + // the closure param `x` is invisible here (opaque nested scope). + expect(bindingIdxs(cfg, 'x')).toHaveLength(0); + }); + + it("multi-line call → the site's `at` line is the call_expression's start line", () => { + const cfg = swift.cfgOf( + `func f(a: Int, b: Int) {\n foo(\n a,\n b\n )\n}\n`, + ); + const s = siteFact(cfg).sites![0]; + expect(s.callee).toBe('foo'); + // `call_expression` starts on line 2 even though the args span lines 3-4. + expect(s.at![0]).toBe(2); + expect(s.at![1]).toBe(4); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + }); + + it('two calls in one function → two distinct call sites', () => { + const cfg = swift.cfgOf(`func f(a: Int, b: Int) {\n foo(a)\n bar(b)\n}\n`); + const calls = allSites(cfg).filter((s) => s.kind === 'call'); + expect(calls.map((s) => s.callee).sort()).toEqual(['bar', 'foo']); + }); + + it('let x = g(y) → resultDefs carries x; arg y occurs at position 0', () => { + const cfg = swift.cfgOf(`func f(y: Int) {\n let x = g(y)\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('g'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'y')]]); + // `at` is the call_expression start `g` (col 12, after ` let x = `). + expect(s.at).toEqual([2, 12]); + }); + + it('x = g(y) assignment → resultDefs carries x (plain = scalar lvalue)', () => { + const cfg = swift.cfgOf(`func f(y: Int) {\n var x = 0\n x = g(y)\n}\n`); + const s = siteFact(cfg, 3).sites![0]; + expect(s.callee).toBe('g'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'y')]]); + }); + + it('init call let u = User(name: n) → kind call (Swift has no `new`)', () => { + const cfg = swift.cfgOf(`func f(n: String) {\n let u = User(name: n)\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('User'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'u')]); + // the labeled value `n` occurs at position 0 (the `name:` label is dropped). + expect(s.args).toEqual([[bindingIdx(cfg, 'n')]]); + // anchored on the call_expression start `User` (col 12) — matches the Swift + // `@reference.call.constructor` anchor on the same node. + expect(s.at).toEqual([2, 12]); + }); + + it('nested call exec(escape(x)) → inner site parent-linked, occurrence via-tagged', () => { + const cfg = swift.cfgOf(`func f(x: Int) {\n exec(escape(x))\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(2); + const execIdx = sites.findIndex((s) => s.callee === 'exec'); + const escapeIdx = sites.findIndex((s) => s.callee === 'escape'); + const x = bindingIdx(cfg, 'x'); + expect(sites[escapeIdx].args).toEqual([[x]]); + expect(sites[escapeIdx].parent).toEqual([execIdx, 0]); + expect(sites[execIdx].args).toEqual([[[x, escapeIdx]]]); + }); + + it('labeled argument foo(name: v) → only the value v is an occurrence (label is not a use)', () => { + const cfg = swift.cfgOf(`func f(v: Int) {\n foo(name: v)\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('foo'); + // `name` mints no binding; `v` occurs at position 0. + expect(bindingIdxs(cfg, 'name')).toHaveLength(0); + expect(s.args).toEqual([[bindingIdx(cfg, 'v')]]); + }); + + it('let b = req.body value read → a member-read site (no call), object = req', () => { + const cfg = swift.cfgOf(`func f(req: Req) {\n let b = req.body\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const read = sites.find((s) => s.kind === 'member-read')!; + expect(read.object).toBe(bindingIdx(cfg, 'req')); + expect(read.property).toBe('body'); + expect(sites.filter((s) => s.kind === 'call')).toHaveLength(0); + }); + + it('def/use facts stay intact alongside the new sites (regression guard)', () => { + const cfg = swift.cfgOf(`func f(y: Int) {\n let x = g(y)\n use(x)\n}\n`); + const x = bindingIdx(cfg, 'x'); + const y = bindingIdx(cfg, 'y'); + // the binding table + def/use are coherent: x is defined, y and x are used. + expect(defsOf(cfg)).toContain(x); + expect(usesOf(cfg)).toContain(y); + expect(usesOf(cfg)).toContain(x); + expect(cfg.bindings![y].kind).toBe('param'); + }); +}); + +// ── Rust call-site harvest ─────────────────────────────────────────────────── +// Drives the REAL Rust CFG visitor (RustHarvester) against real source via the +// language-agnostic harness, mirroring the Swift / Kotlin / Python site tests +// above. Rust has ONE call node, `call_expression { function, arguments }`, whose +// `function` takes three shapes: a bare `identifier` (free `foo(x)`), a +// `field_expression` (method `a.method(x)` — `.` access, dotted callee + root +// receiver), and a `scoped_identifier` (path `Foo::bar(x)` / `a::b::c(x)` — the +// `::` segments joined with `.` so the LEAF is the tail, `Foo::bar` ⇒ leaf +// `bar`, matching the CALLS `@reference.name` tail capture AND calleesOfBlock's +// `lastIndexOf('.')` leaf rule). The turbofish `foo::(x)` (`generic_function`) +// unwraps to the same site as `foo(x)`. Macros (`println!(…)`) are +// `macro_invocation`, NOT `call_expression`, and are tagged `@reference.macro` +// (a disjoint namespace) — NOT a call — so the harvester records NO site for +// them (its arg idents still walk for uses). Rust has no `new` — every site is +// `kind: 'call'`. The `at` anchor is byte-aligned with the Rust +// `@reference.call.free/.member/.constructor` CALLS anchor, which the scope query +// (captures.ts) places on the WHOLE `call_expression` node (the Swift/Go/Python/ +// Kotlin whole-call model, NOT Dart's callee-name model) — so for a member / +// chained / path call `at` starts at the call's head segment (the receiver +// `a` / the path head `Foo`), exactly where the CALLS anchor starts. Each `at` +// below was confirmed byte-equal to the real `emitRustScopeCaptures` +// `@reference.call.*` atRange. The grammar is loaded the same way the Rust +// visitor test loads it (createRequire('tree-sitter-rust')). + +import { createRustCfgVisitor } from '../../../src/core/ingestion/cfg/visitors/rust.js'; + +const rustGrammar = createRequire(import.meta.url)('tree-sitter-rust') as Parameters< + typeof makeCfgHarness +>[0]; +const rust: CfgHarness = makeCfgHarness(rustGrammar, createRustCfgVisitor(), 'fixture.rs'); + +describe('Rust call-site harvest', () => { + it('foo(a, b) → one call site, positions 0→[a], 1→[b], at the call_expression line/col', () => { + const cfg = rust.cfgOf(`fn f(a: i32, b: i32) {\n foo(a, b);\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('foo'); + expect(s.receiver).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + expect(s.parent).toBeUndefined(); + // `at` is the `call_expression` node start: line 2 (1-based), col 4 (indent), + // byte-equal to the Rust `@reference.call.free` atRange [2,4]. + expect(s.at).toEqual([2, 4]); + }); + + it('a.method(x) → dotted callee path + receiver = a, at the call_expression start', () => { + const cfg = rust.cfgOf(`fn f(a: Foo, x: i32) {\n a.method(x);\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.callee).toBe('a.method'); + expect(s.receiver).toBe(bindingIdx(cfg, 'a')); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + // chain-length-1 callee: the access IS the callee — no member-read site, and + // the receiver use is recorded exactly once. + expect(siteFact(cfg, 2).uses.filter((u) => u === bindingIdx(cfg, 'a'))).toHaveLength(1); + // `at` is the call_expression start (the receiver `a`), col 4 — NOT the method + // name. The Rust `@reference.call.member` anchor keys on the same node ([2,4]). + expect(s.at).toEqual([2, 4]); + }); + + it('a.b.c() → callee path a.b.c, receiver = root a, plus a mid-chain member read', () => { + const cfg = rust.cfgOf(`fn f(a: Foo) {\n a.b.c();\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const call = sites.find((s) => s.kind === 'call')!; + expect(call.callee).toBe('a.b.c'); + expect(call.receiver).toBe(bindingIdx(cfg, 'a')); + // `at` is the whole call_expression start (the root `a`), col 4. + expect(call.at).toEqual([2, 4]); + // the innermost access `a.b` (the non-callee load) is a member read. + const read = sites.find((s) => s.kind === 'member-read')!; + expect(read.object).toBe(bindingIdx(cfg, 'a')); + expect(read.property).toBe('b'); + }); + + it('Foo::bar(x) path call → callee Foo.bar (leaf bar), no receiver (Foo is a type)', () => { + const cfg = rust.cfgOf(`fn f(x: i32) {\n Foo::bar(x);\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.kind).toBe('call'); + // `::` joined with `.` so the leaf (after last `.`) is the tail `bar` — the + // SAME leaf the CALLS `@reference.name` tail-identifier capture produces, so + // calleesOfBlock's `lastIndexOf('.')` slice yields `bar` and the name-fallback + // stays correct. + expect(s.callee).toBe('Foo.bar'); + expect(s.callee!.slice(s.callee!.lastIndexOf('.') + 1)).toBe('bar'); + // `Foo` is a type/module head — no value binding — so no receiver. + expect(s.receiver).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + // anchored on the call_expression start (`Foo`, col 4) — matches the Rust + // `@reference.call.free` atRange [2,4] for the scoped form. + expect(s.at).toEqual([2, 4]); + }); + + it('a::b::c(x) path call with a LOCAL head → callee a.b.c (leaf c), receiver = a', () => { + const cfg = rust.cfgOf(`fn f(a: A, x: i32) {\n a::b::c(x);\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + // multi-segment scoped path → joined with `.`; leaf after last `.` is `c`. + expect(s.callee).toBe('a.b.c'); + expect(s.callee!.slice(s.callee!.lastIndexOf('.') + 1)).toBe('c'); + // the head segment `a` IS a bound local (a param), so it is the receiver. + expect(s.receiver).toBe(bindingIdx(cfg, 'a')); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + expect(s.at).toEqual([2, 4]); + }); + + it('turbofish foo::(x) → unwraps generic_function to the same site as foo(x)', () => { + const cfg = rust.cfgOf(`fn f(x: i32) {\n foo::(x);\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('foo'); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + expect(s.at).toEqual([2, 4]); + }); + + it('chained a.b().c() → two distinct call sites (inner a.b, outer call-rooted)', () => { + const cfg = rust.cfgOf(`fn f(a: Foo) {\n a.b().c();\n}\n`); + const calls = siteFact(cfg, 2).sites!.filter((s) => s.kind === 'call'); + expect(calls).toHaveLength(2); + // the inner `a.b()` is a member call with receiver = root a. + const inner = calls.find((s) => s.callee === 'a.b')!; + expect(inner.receiver).toBe(bindingIdx(cfg, 'a')); + // the outer `(a.b()).c()` is rooted on a CALL result — no static callee path / + // receiver — but it is still its own call site, anchored on the same line. + const outer = calls.find((s) => s.callee === undefined)!; + expect(outer.kind).toBe('call'); + expect(outer.receiver).toBeUndefined(); + expect(outer.at).toEqual([2, 4]); + }); + + it("multi-line call → the site's `at` line is the call_expression's start line", () => { + const cfg = rust.cfgOf(`fn f(a: i32, b: i32) {\n foo(\n a,\n b,\n );\n}\n`); + const s = siteFact(cfg).sites![0]; + expect(s.callee).toBe('foo'); + // `call_expression` starts on line 2 even though the args span lines 3-4. + expect(s.at![0]).toBe(2); + expect(s.at![1]).toBe(4); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + }); + + it('two calls in one function → two distinct call sites', () => { + const cfg = rust.cfgOf(`fn f(a: i32, b: i32) {\n foo(a);\n bar(b);\n}\n`); + const calls = allSites(cfg).filter((s) => s.kind === 'call'); + expect(calls.map((s) => s.callee).sort()).toEqual(['bar', 'foo']); + }); + + it('let r = g(x) → resultDefs carries r; arg x occurs at position 0', () => { + const cfg = rust.cfgOf(`fn f(x: i32) {\n let r = g(x);\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('g'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'r')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + // `at` is the call_expression start `g` (col 12, after ` let r = `). + expect(s.at).toEqual([2, 12]); + }); + + it('r = g(x) plain assignment → resultDefs carries r (scalar lvalue)', () => { + const cfg = rust.cfgOf(`fn f(x: i32) {\n let mut r = 0;\n r = g(x);\n}\n`); + const s = siteFact(cfg, 3).sites![0]; + expect(s.callee).toBe('g'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'r')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + }); + + it('foo(x)? try-call → still one call site (the ? wraps the call_expression)', () => { + const cfg = rust.cfgOf(`fn f(x: i32) -> Result<(), ()> {\n foo(x)?;\n Ok(())\n}\n`); + const s = siteFact(cfg, 2).sites![0]; + expect(s.callee).toBe('foo'); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + expect(s.at).toEqual([2, 4]); + }); + + it('nested call exec(escape(x)) → inner site parent-linked, occurrence via-tagged', () => { + const cfg = rust.cfgOf(`fn f(x: i32) {\n exec(escape(x));\n}\n`); + const sites = siteFact(cfg, 2).sites!; + expect(sites).toHaveLength(2); + const execIdx = sites.findIndex((s) => s.callee === 'exec'); + const escapeIdx = sites.findIndex((s) => s.callee === 'escape'); + const x = bindingIdx(cfg, 'x'); + expect(sites[escapeIdx].args).toEqual([[x]]); + expect(sites[escapeIdx].parent).toEqual([execIdx, 0]); + expect(sites[execIdx].args).toEqual([[[x, escapeIdx]]]); + }); + + it('println!(...) macro → NO call site recorded (disjoint @reference.macro namespace)', () => { + const cfg = rust.cfgOf(`fn f(x: i32) {\n println!("{}", x);\n}\n`); + // A macro is not a call_expression and is resolved via the MacroRegistry, NOT + // CALLS — so the harvester records no site (no spurious `println` callee), but + // the macro's argument identifier `x` still walks for a use. + expect(allSites(cfg)).toHaveLength(0); + expect(usesOf(cfg)).toContain(bindingIdx(cfg, 'x')); + }); + + it('a.b field read (no call) → a member-read site, object = a', () => { + const cfg = rust.cfgOf(`fn f(a: Foo) {\n let _v = a.b;\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const read = sites.find((s) => s.kind === 'member-read')!; + expect(read.object).toBe(bindingIdx(cfg, 'a')); + expect(read.property).toBe('b'); + expect(sites.filter((s) => s.kind === 'call')).toHaveLength(0); + }); + + it('def/use facts stay intact alongside the new sites (regression guard)', () => { + const cfg = rust.cfgOf(`fn f(y: i32) {\n let x = g(y);\n use(x);\n}\n`); + const x = bindingIdx(cfg, 'x'); + const y = bindingIdx(cfg, 'y'); + // the binding table + def/use are coherent: x is defined, y and x are used. + expect(defsOf(cfg)).toContain(x); + expect(usesOf(cfg)).toContain(y); + expect(usesOf(cfg)).toContain(x); + expect(cfg.bindings![y].kind).toBe('param'); + }); + + // ── struct-literal constructors (U4) ────────────────────────────────────── + // A struct literal `Point { x: 1 }` is a `struct_expression`, NOT a + // `call_expression`; the Rust CALLS query tags it `@reference.call.constructor`, + // so the harvester records a `kind: 'new'` site whose callee leaf is the struct + // TYPE tail (so the resolved constructor id joins into `calleeIds`). The `at` is + // the `struct_expression` start — byte-equal to the + // `@reference.call.constructor` atRange (verified byte-exact for plain / scoped / + // turbofish forms; all anchor on col 12 after ` let p = `). + + it('let p = Point { x: 1, y: 2 } → one kind:new site, callee Point, at the struct_expression start', () => { + const cfg = rust.cfgOf(`fn f() {\n let p = Point { x: 1, y: 2 };\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const newSites = sites.filter((s) => s.kind === 'new'); + expect(newSites).toHaveLength(1); + const s = newSites[0]; + expect(s.kind).toBe('new'); + expect(s.callee).toBe('Point'); + // a bare type head is no value binding ⇒ no receiver. + expect(s.receiver).toBeUndefined(); + // `at` is the `struct_expression` start (col 12, after ` let p = `), byte-equal + // to the Rust `@reference.call.constructor` atRange [2,12]. + expect(s.at).toEqual([2, 12]); + }); + + it('scoped mymod::Point { x: 1 } → callee path mymod.Point whose leaf is Point', () => { + const cfg = rust.cfgOf(`fn f() {\n let p = mymod::Point { x: 1 };\n}\n`); + const s = siteFact(cfg, 2).sites!.find((x) => x.kind === 'new')!; + // `::` joined with `.` so the leaf (after last `.`) is the tail `Point` — the + // SAME tail the CALLS `@reference.name` capture resolves, so calleesOfBlock's + // `lastIndexOf('.')` slice yields `Point`. + expect(s.callee).toBe('mymod.Point'); + expect(s.callee!.slice(s.callee!.lastIndexOf('.') + 1)).toBe('Point'); + expect(s.receiver).toBeUndefined(); + expect(s.at).toEqual([2, 12]); + }); + + it('turbofish Foo:: { x: 1 } → callee Foo (turbofish args dropped), at the struct start', () => { + const cfg = rust.cfgOf(`fn f() {\n let p = Foo:: { x: 1 };\n}\n`); + const s = siteFact(cfg, 2).sites!.find((x) => x.kind === 'new')!; + expect(s.callee).toBe('Foo'); + expect(s.at).toEqual([2, 12]); + }); + + it('Point { x: f() } → the struct site PLUS the inner f() call site, x value recorded', () => { + const cfg = rust.cfgOf(`fn f(y: i32) {\n let p = Point { x: f(), y };\n}\n`); + const sites = siteFact(cfg, 2).sites!; + const structSite = sites.find((s) => s.kind === 'new')!; + const callSite = sites.find((s) => s.kind === 'call')!; + // the struct site is recorded as a constructor, the inner `f()` as its own call. + expect(structSite.callee).toBe('Point'); + expect(callSite.callee).toBe('f'); + // the inner `f()` is the value of field `x` (position 0) — parent-linked to the + // struct site, so the field value is tracked, not the field NAME. + const structIdx = sites.indexOf(structSite); + expect(callSite.parent).toEqual([structIdx, 0]); + // the shorthand field `y` (position 1) records the local `y` as a value use. + expect(structSite.args).toEqual([[], [bindingIdx(cfg, 'y')]]); + expect(usesOf(cfg)).toContain(bindingIdx(cfg, 'y')); + }); + + it('struct literal def/use facts stay intact (regression guard)', () => { + const cfg = rust.cfgOf(`fn f(a: i32) {\n let p = Point { x: a };\n}\n`); + const a = bindingIdx(cfg, 'a'); + const p = bindingIdx(cfg, 'p'); + // `p` is defined by the `let`; `a` (the field value) is used; field name `x` is not. + expect(defsOf(cfg)).toContain(p); + expect(usesOf(cfg)).toContain(a); + expect(cfg.bindings!.map((b) => b.name)).not.toContain('x'); + }); +}); diff --git a/gitnexus/test/unit/cfg/memoized-reaching-defs.test.ts b/gitnexus/test/unit/cfg/memoized-reaching-defs.test.ts new file mode 100644 index 000000000..7bdaea3ca --- /dev/null +++ b/gitnexus/test/unit/cfg/memoized-reaching-defs.test.ts @@ -0,0 +1,36 @@ +// U12 — the per-file memoized reaching-defs solver must be a TRANSPARENT cache: +// byte-identical to computeReachingDefs, one solve per (cfg, limits) bucket, and +// the limits ARE part of the key (so the RD-emit bucket, which passes +// maxBlockVisits, never aliases the harvest/taint bucket, which does not). + +import { describe, it, expect } from 'vitest'; +import { + computeReachingDefs, + createMemoizedReachingDefs, +} from '../../../src/core/ingestion/cfg/reaching-defs.js'; +import { cfgOf } from '../../helpers/ts-cfg-harness.js'; + +describe('createMemoizedReachingDefs (U12)', () => { + it('caches by (cfg, limits) — a repeat call returns the same result, solved once', () => { + const cfg = cfgOf(`function f(a: string) { const b = a; return b; }`); + const solve = createMemoizedReachingDefs(); + const first = solve(cfg, { maxFacts: 100 }); + const second = solve(cfg, { maxFacts: 100 }); + expect(second).toBe(first); + }); + + it('matches computeReachingDefs byte-for-byte (transparent)', () => { + const cfg = cfgOf(`function f(a: string, b: string) { const c = a; return c; }`); + const solve = createMemoizedReachingDefs(); + expect(solve(cfg, { maxFacts: 100 })).toEqual(computeReachingDefs(cfg, { maxFacts: 100 })); + }); + + it('keys on limits — maxBlockVisits is a SEPARATE bucket (emit vs harvest split)', () => { + const cfg = cfgOf(`function f(a: string) { return a; }`); + const solve = createMemoizedReachingDefs(); + const withVisits = solve(cfg, { maxFacts: 100, maxBlockVisits: 64 }); + const withoutVisits = solve(cfg, { maxFacts: 100 }); + expect(solve(cfg, { maxFacts: 100, maxBlockVisits: 64 })).toBe(withVisits); + expect(solve(cfg, { maxFacts: 100 })).toBe(withoutVisits); + }); +}); diff --git a/gitnexus/test/unit/cfg/python-visitor.test.ts b/gitnexus/test/unit/cfg/python-visitor.test.ts index 473cb6d4a..76411af39 100644 --- a/gitnexus/test/unit/cfg/python-visitor.test.ts +++ b/gitnexus/test/unit/cfg/python-visitor.test.ts @@ -380,12 +380,13 @@ describe('Python CfgVisitor — production CDG probe (plan-required)', () => { }); }); -describe('Python CfgVisitor — no taint sites harvested (this unit)', () => { - it('statements carry NO sites key (taint substrate is a later step)', () => { +describe('Python CfgVisitor — taint call sites harvested (this unit)', () => { + it('call statements now carry harvested call sites (callee path recorded)', () => { const cfg = py.cfgOf(`def f(cmd):\n exec(cmd)\n x = escape(cmd)\n use(x)\n`); - const anySites = cfg.blocks.some((b) => - (b.statements ?? []).some((s) => (s as { sites?: unknown }).sites !== undefined), - ); - expect(anySites).toBe(false); + const callees = cfg.blocks + .flatMap((b) => b.statements ?? []) + .flatMap((s) => s.sites ?? []) + .map((site) => site.callee); + expect(callees).toEqual(expect.arrayContaining(['exec', 'escape', 'use'])); }); }); diff --git a/gitnexus/test/unit/cfg/reaching-def-reason-codec.test.ts b/gitnexus/test/unit/cfg/reaching-def-reason-codec.test.ts new file mode 100644 index 000000000..b3c263385 --- /dev/null +++ b/gitnexus/test/unit/cfg/reaching-def-reason-codec.test.ts @@ -0,0 +1,127 @@ +import { describe, it, expect } from 'vitest'; +import { + encodeReachingDefReason, + encodeReachingDefReasonPairs, + decodeReachingDefReason, + REACHING_DEF_REASON_CODEC_VERSION, +} from '../../../src/core/ingestion/cfg/reaching-def-reason-codec.js'; +import { sanitizeUTF8, escapeCSVField } from '../../../src/core/lbug/csv-generator.js'; + +// FU-B-2: the REACHING_DEF `reason` codec carries the binding NAME (first, +// verbatim — `pdg_query` flows keys on it) plus a compact versioned annotation +// holding the ORDERED LIST of def/use-line pairs for the (block-pair, binding) +// group. The statement-granular intra-block projection (pdg-impact.ts) walks the +// annotation's def→use pairs to recover a coalesced block's interior dependents, +// so the round-trip + CSV-survival contract is load-bearing. + +describe('FU-B-2 reaching-def reason codec', () => { + it('round-trips name + a single def/use pair', () => { + const wire = encodeReachingDefReason('acc', 7, 9); + expect(wire).toBe(`acc|${REACHING_DEF_REASON_CODEC_VERSION}:7:9`); + expect(decodeReachingDefReason(wire)).toEqual({ + name: 'acc', + pairs: [{ defLine: 7, useLine: 9 }], + defLine: 7, + useLine: 9, + }); + }); + + it('round-trips a MULTI-pair list (same-binding reassignment chain)', () => { + // `acc = f(acc); acc = g(acc); acc = h(acc)` — one (self-block, accIdx) group, + // three def→use steps. The full list must survive so the projection walk can + // chain 24->25->26->27 to fixpoint (a first-pair-only encoding would stop at 25). + const wire = encodeReachingDefReasonPairs('acc', [ + { defLine: 24, useLine: 25 }, + { defLine: 25, useLine: 26 }, + { defLine: 26, useLine: 27 }, + ]); + expect(wire).toBe(`acc|${REACHING_DEF_REASON_CODEC_VERSION}:24:25;25:26;26:27`); + expect(decodeReachingDefReason(wire)).toEqual({ + name: 'acc', + pairs: [ + { defLine: 24, useLine: 25 }, + { defLine: 25, useLine: 26 }, + { defLine: 26, useLine: 27 }, + ], + // `defLine`/`useLine` mirror the FIRST pair for back-compat consumers. + defLine: 24, + useLine: 25, + }); + }); + + it('decodes a self-edge (defLine === useLine) — a same-line read-then-write', () => { + const wire = encodeReachingDefReason('x', 12, 12); + expect(decodeReachingDefReason(wire)).toEqual({ + name: 'x', + pairs: [{ defLine: 12, useLine: 12 }], + defLine: 12, + useLine: 12, + }); + }); + + it('preserves a synthetic-binding name (`name@module`) verbatim — `@` is not the separator', () => { + const wire = encodeReachingDefReason('config@module', 3, 40); + expect(decodeReachingDefReason(wire)).toEqual({ + name: 'config@module', + pairs: [{ defLine: 3, useLine: 40 }], + defLine: 3, + useLine: 40, + }); + }); + + it('a legacy bare-name reason (no annotation) decodes to the name with no pairs', () => { + expect(decodeReachingDefReason('total')).toEqual({ name: 'total', pairs: [] }); + }); + + it('drops a malformed/negative line to the bare name (sound default — never a bad annotation)', () => { + expect(encodeReachingDefReason('y', -1, 9)).toBe('y'); + expect(encodeReachingDefReason('y', 7, Number.NaN)).toBe('y'); + expect(decodeReachingDefReason(encodeReachingDefReason('y', -1, 9))).toEqual({ + name: 'y', + pairs: [], + }); + }); + + it('drops a single malformed pair but keeps the well-formed pairs of the list', () => { + // `|1:1:2;bad;5:6` — the middle chunk is not a `def:use` pair; it is + // dropped, the rest kept (the decoder never throws and never loses good pairs). + expect(decodeReachingDefReason('w|1:1:2;bad;5:6')).toEqual({ + name: 'w', + pairs: [ + { defLine: 1, useLine: 2 }, + { defLine: 5, useLine: 6 }, + ], + defLine: 1, + useLine: 2, + }); + }); + + it('never throws on garbage — yields the best-effort name, no pairs', () => { + expect(decodeReachingDefReason(undefined)).toEqual({ name: '', pairs: [] }); + expect(decodeReachingDefReason(123)).toEqual({ name: '', pairs: [] }); + // wrong version / shape → name kept, annotation discarded + expect(decodeReachingDefReason('z|9:1:2')).toEqual({ name: 'z', pairs: [] }); + expect(decodeReachingDefReason('z|1:abc:2')).toEqual({ name: 'z', pairs: [] }); + expect(decodeReachingDefReason('z|1:2')).toEqual({ name: 'z', pairs: [] }); + }); + + it('survives escapeCSVField ∘ sanitizeUTF8 byte-exact (printable ASCII only)', () => { + const wire = encodeReachingDefReasonPairs('count', [ + { defLine: 100, useLine: 250 }, + { defLine: 250, useLine: 400 }, + ]); + expect(sanitizeUTF8(wire)).toBe(wire); + // round-trips out of the escaped CSV cell (strip the surrounding quotes) + const escaped = escapeCSVField(wire); + const unquoted = escaped.slice(1, -1).replace(/""/g, '"'); + expect(decodeReachingDefReason(unquoted)).toEqual({ + name: 'count', + pairs: [ + { defLine: 100, useLine: 250 }, + { defLine: 250, useLine: 400 }, + ], + defLine: 100, + useLine: 250, + }); + }); +}); diff --git a/gitnexus/test/unit/cfg/ruby-visitor.test.ts b/gitnexus/test/unit/cfg/ruby-visitor.test.ts index 56d254925..c304dca86 100644 --- a/gitnexus/test/unit/cfg/ruby-visitor.test.ts +++ b/gitnexus/test/unit/cfg/ruby-visitor.test.ts @@ -412,12 +412,17 @@ describe('Ruby CfgVisitor — production CDG probe (plan-required)', () => { }); }); -describe('Ruby CfgVisitor — no taint sites harvested (this unit)', () => { - it('statements carry NO sites key (taint substrate is a later step)', () => { +describe('Ruby CfgVisitor — call sites ARE harvested (this unit)', () => { + it('statements carry a sites key for each call (see harvest.test.ts for the shapes)', () => { const cfg = rb.cfgOf(`def f(cmd)\n exec(cmd)\n x = escape(cmd)\n use(x)\nend\n`); - const anySites = cfg.blocks.some((b) => - (b.statements ?? []).some((s) => (s as { sites?: unknown }).sites !== undefined), - ); - expect(anySites).toBe(false); + const callees = cfg.blocks + .flatMap((b) => b.statements ?? []) + .flatMap((s) => s.sites ?? []) + .filter((site) => site.kind === 'call') + .map((site) => site.callee); + // `exec`, `escape`, and `use` each open a call site (the taint substrate). + expect(callees).toContain('exec'); + expect(callees).toContain('escape'); + expect(callees).toContain('use'); }); }); diff --git a/gitnexus/test/unit/cli-impact-pdg-format.test.ts b/gitnexus/test/unit/cli-impact-pdg-format.test.ts new file mode 100644 index 000000000..8bfe49ed4 --- /dev/null +++ b/gitnexus/test/unit/cli-impact-pdg-format.test.ts @@ -0,0 +1,558 @@ +/** + * U5 — CLI / consumer rendering for PDG (`mode:'pdg'`) impact results. + * + * Guards the KTD8 presentation contract: PDG results must render HONESTLY — + * - inter-procedural symbol reach under a neutral heading, NOT callgraph severity labels; + * - the unified PDG caveat (statement reach in affectedStatements, symbol reach in interproceduralByDepth/byDepth); + * - degradation → the "run analyze --pdg" remediation, NOT a zero blast radius; + * - no-body (KTD6) → the "not applicable to this symbol kind" caveat, NOT + * "isolated / no dependencies"; + * - the callgraph DI/dynamic-dispatch epistemic copy is NEVER printed for PDG. + * + * And the standing interchangeability contract (KTD8): `mode:'callgraph'` + * rendering stays byte-identical (regression guard). + */ +import { describe, expect, it } from 'vitest'; +import { formatImpactResult, getNextStepHint } from '../../src/cli/eval-server.js'; + +// A representative PDG findings result, shaped exactly like +// `assemblePdgImpactResult` (pdg-impact.ts) emits. +function pdgFindings(overrides: Record = {}): Record { + const items = [ + { + depth: 1, + id: 'Function:src/svc.ts:applyDiscount', + name: 'applyDiscount', + type: 'Function', + filePath: 'src/svc.ts', + processes: [], + }, + { + depth: 1, + id: 'Function:src/svc.ts:finalizeTotal', + name: 'finalizeTotal', + type: 'Function', + filePath: 'src/svc.ts', + processes: [], + }, + ]; + return { + mode: 'pdg', + pdgResultVersion: 1, + target: { + id: 'Function:src/svc.ts:computeTotal', + name: 'computeTotal', + type: 'Function', + filePath: 'src/svc.ts', + }, + direction: 'downstream', + impactedCount: 2, + risk: 'UNKNOWN', + epistemic: 'pdg-intra-procedural', + note: + "mode:'pdg' — intra-procedural Program Dependence Graph. 2 owning symbols reached via 4 " + + 'dependence blocks (downstream over CDG + REACHING_DEF). Inter-procedural symbol reach ' + + 'is included using the resolved symbol graph; statement-level PDG reach remains in affectedStatements.', + reachableBlocks: ['b1', 'b2', 'b3', 'b4'], + blockCount: 4, + depthReached: 2, + unresolvedBlockCount: 0, + ambiguousProjectionCount: 0, + summary: { direct: 2, processes_affected: 0, modules_affected: 0 }, + byDepthCounts: { 1: 2 }, + affected_processes: [], + affected_modules: [], + byDepth: { 1: items }, + ...overrides, + }; +} + +describe('formatImpactResult — PDG (mode:pdg) rendering', () => { + it('renders a PDG ambiguous target without fabricated zero blast-radius counts', () => { + const out = formatImpactResult({ + status: 'ambiguous', + mode: 'pdg', + message: + "Found 2 symbols matching 'login'. Disambiguate with target_uid for a single authoritative PDG result.", + target: { name: 'login' }, + direction: 'upstream', + totalCandidates: 2, + impactedCount: 0, + risk: 'UNKNOWN', + candidates: [ + { + uid: 'func:login:1', + name: 'login', + kind: 'Function', + filePath: 'src/auth.ts', + line: 5, + score: 1, + }, + { + uid: 'func:login:2', + name: 'login', + kind: 'Function', + filePath: 'src/admin/login.ts', + line: 8, + score: 0.91, + }, + ], + }); + + expect(out).toContain('login: AMBIGUOUS'); + expect(out).toContain('PDG impact was not computed'); + expect(out).toContain('func:login:1'); + expect(out).not.toContain('Max blast radius 0'); + expect(out).not.toContain('[0 upstream'); + }); + + it('renders unified PDG symbol reach without callgraph severity labels', () => { + const out = formatImpactResult(pdgFindings()); + + // PDG framing — NOT the callgraph "depth N / WILL BREAK (direct)" labels. + expect(out).toContain('Inter-procedural symbol reach'); + expect(out).toContain('d=1 (2)'); + expect(out).not.toContain('WILL BREAK (direct)'); + expect(out).not.toContain('LIKELY AFFECTED'); + expect(out).not.toContain('MAY NEED TESTING'); + // The callgraph "Blast radius for ... will break if changed" headline must + // not leak into PDG output. + expect(out).not.toContain('Blast radius for'); + + // The affected symbols are listed. + expect(out).toContain('applyDiscount'); + expect(out).toContain('finalizeTotal'); + expect(out).toContain('src/svc.ts'); + + // The unified contract is present. + expect(out).toContain('statement-level PDG reach remains in affectedStatements'); + + // The callgraph DI / dynamic-dispatch lower-bound copy must NEVER appear. + expect(out).not.toContain('dynamic dispatch'); + expect(out).not.toContain('binding via DI'); + }); + + it('carries the stable pdgResultVersion discriminator on the findings result', () => { + // The PDG result family advertises a contract version (FIX #2) so external + // MCP/agent consumers can version against future shape evolution. It is a + // mode:'pdg'-only field — never on the default callgraph result. + expect(pdgFindings()).toMatchObject({ mode: 'pdg', pdgResultVersion: 1 }); + }); + + it('surfaces ambiguous-projection and unresolved block counts honestly', () => { + const out = formatImpactResult( + pdgFindings({ + ambiguousProjectionCount: 2, + unresolvedBlockCount: 1, + byDepth: { + 1: [ + { + depth: 1, + id: 'Function:src/svc.ts:applyDiscount', + name: 'applyDiscount', + type: 'Function', + filePath: 'src/svc.ts', + ambiguous: true, + processes: [], + }, + { + depth: 1, + id: null, + name: '(top-level)', + type: 'BasicBlock', + filePath: 'src/svc.ts', + unresolved: true, + processes: [], + }, + ], + }, + byDepthCounts: { 1: 2 }, + }), + ); + expect(out).toContain('2 block(s) could not be attributed'); + expect(out).toContain('1 dependence block(s) map to no owning'); + // The shadow / ambiguous rows carry their flags inline. + expect(out).toContain('[ambiguous]'); + expect(out).toContain('[unresolved]'); + }); + + it('flags truncation honestly', () => { + const out = formatImpactResult(pdgFindings({ truncated: true, truncatedBy: 'depth' })); + expect(out).toContain('Truncated'); + expect(out).toContain('by depth'); + expect(out).toContain('deeper PDG impacts may exist'); + }); + + it('renders multiple truncation causes honestly', () => { + const out = formatImpactResult( + pdgFindings({ + truncated: true, + truncatedBy: 'depth', + truncatedByReasons: ['depth', 'limit'], + }), + ); + expect(out).toContain('Truncated'); + expect(out).toContain('by depth, limit'); + }); + + it('renders the degradation note as remediation, not a zero/empty blast radius', () => { + // Shaped like the `_impactImpl` pdgLayer-degradation early return. + const out = formatImpactResult({ + mode: 'pdg', + pdgLayer: 'no-layer', + note: 'No PDG layer in this index. Run `gitnexus analyze --pdg` to build it.', + target: { name: 'computeTotal' }, + direction: 'downstream', + impactedCount: 0, + risk: 'UNKNOWN', + }); + // The remediation guidance is present. + expect(out).toContain('analyze --pdg'); + expect(out).toContain('no usable PDG layer'); + // It must NOT read as a confident "isolated / no dependencies / safe". + expect(out).not.toContain('isolated'); + expect(out).not.toContain('No downstream dependencies found'); + }); + + it('names the missing sub-layer in a partial-degradation note', () => { + const out = formatImpactResult({ + mode: 'pdg', + pdgLayer: 'sub-layer-missing', + missingSubLayer: 'REACHING_DEF', + note: 'CDG present but REACHING_DEF absent.', + target: { name: 'computeTotal' }, + direction: 'downstream', + impactedCount: 0, + risk: 'UNKNOWN', + }); + expect(out).toContain('REACHING_DEF'); + expect(out).toContain('analyze --pdg'); + expect(out).not.toContain('isolated'); + }); + + it('renders the no-body (KTD6) caveat, not "isolated / no dependencies"', () => { + // Shaped like `_runImpactPDG`'s no-body early return. + const out = formatImpactResult({ + mode: 'pdg', + target: { + id: 'Interface:src/types.ts:Card', + name: 'Card', + type: 'Interface', + filePath: 'src/types.ts', + }, + direction: 'downstream', + reachableBlocks: [], + blockCount: 0, + truncated: false, + depthReached: 0, + epistemic: 'no-pdg-body', + note: + "'Card' has no PDG body — no BasicBlocks / control- or data-dependence edges exist for " + + 'this symbol (e.g. an interface, type alias, abstract/ambient member, or a one-line ' + + 'declaration with no CFG). This is NOT a confident "no impact": the local PDG ' + + 'statement slice cannot model this symbol kind. Inter-procedural symbol reach may still be attached.', + impactedCount: 0, + risk: 'UNKNOWN', + byDepth: {}, + byDepthCounts: { 1: 0 }, + summary: { direct: 0, processes_affected: 0, modules_affected: 0 }, + affected_processes: [], + affected_modules: [], + unresolvedBlockCount: 0, + ambiguousProjectionCount: 0, + }); + expect(out).toContain('no PDG body'); + expect(out.toLowerCase()).toContain('not applicable'); + // NOT the false-safe callgraph "isolated" headline. (The note may DISCLAIM + // "no impact", but the confident standalone "appears isolated." sentence + // and the callgraph "No ... dependencies found." headline must be absent.) + expect(out).not.toContain('appears isolated'); + expect(out).not.toContain('No downstream dependencies found'); + }); + + it('renders whole-symbol-empty as not-isolated, steering the caller to line:', () => { + // `_runImpactPDG` reachableBlocks.length === 0 path WITHOUT a line (whole- + // symbol seed). The note now frames it as a structurally-empty WHOLE-SYMBOL + // slice and steers to `line:` (the useful statement-anchored mode). + const out = formatImpactResult({ + mode: 'pdg', + target: { + id: 'Function:src/svc.ts:noop', + name: 'noop', + type: 'Function', + filePath: 'src/svc.ts', + }, + direction: 'downstream', + impactedCount: 0, + risk: 'UNKNOWN', + epistemic: 'pdg-intra-procedural', + note: + "'noop' has a PDG body but a WHOLE-SYMBOL downstream slice is empty: " + + 'intra-procedural dependence stays inside the function, so every reachable block ' + + 'is already part of the seed. Pass line: to slice from a specific statement ' + + '(what depends on the code at that line). Inter-procedural symbol reach is attached ' + + 'separately by the unified impact dispatcher.', + reachableBlocks: [], + blockCount: 0, + affectedStatements: [], + affectedStatementCount: 0, + depthReached: 1, + unresolvedBlockCount: 0, + ambiguousProjectionCount: 0, + byDepth: {}, + byDepthCounts: { 1: 0 }, + summary: { direct: 0, processes_affected: 0, modules_affected: 0 }, + affected_processes: [], + affected_modules: [], + }); + expect(out).toContain('no inter-procedural symbols reached'); + // The new note steers to the statement-anchored mode. + expect(out).toContain('WHOLE-SYMBOL'); + expect(out).toMatch(/line:/); + // The caveat may reference the word "isolated" to disclaim it, but the + // confident callgraph "appears isolated." headline must be absent. + expect(out).not.toContain('appears isolated'); + expect(out).not.toContain('No downstream dependencies found'); + expect(out).toContain('Inter-procedural symbol reach is attached'); + }); + + // ── Statement-anchored (mode:'pdg' + line) rendering ────────────────────── + // A representative statement-mode result, shaped like `assemblePdgImpactResult` + // emits when seeded on a line: criterionLine + affectedStatements + count. + function pdgStatementSlice(overrides: Record = {}): Record { + return { + mode: 'pdg', + target: { + id: 'Function:src/svc.ts:accum', + name: 'accum', + type: 'Function', + filePath: 'src/svc.ts', + }, + direction: 'downstream', + criterionLine: 8, + affectedStatements: [ + { line: 10, filePath: 'src/svc.ts', text: 'sum = sum + x;' }, + { line: 12, filePath: 'src/svc.ts', text: 'return sum;' }, + ], + affectedStatementCount: 2, + impactedCount: 0, + risk: 'UNKNOWN', + epistemic: 'pdg-intra-procedural', + note: + "mode:'pdg' — intra-procedural slice from line 8 of 'accum'. 2 statements are " + + 'downstream-dependent on it (over CDG + REACHING_DEF). Inter-procedural symbol reach ' + + 'is attached separately by the unified impact dispatcher.', + reachableBlocks: ['b1', 'b2'], + blockCount: 2, + depthReached: 2, + unresolvedBlockCount: 0, + ambiguousProjectionCount: 0, + summary: { direct: 0, processes_affected: 0, modules_affected: 0 }, + byDepthCounts: {}, + affected_processes: [], + affected_modules: [], + byDepth: {}, + ...overrides, + }; + } + + it('renders a statement slice as an L: list under the criterion-line heading', () => { + const out = formatImpactResult(pdgStatementSlice()); + // Heading carries direction + file:criterionLine + count. + expect(out).toContain('Statements downstream-dependent on src/svc.ts:8 (2):'); + // Each dependent statement renders as ` L: `. + expect(out).toContain(' L10: sum = sum + x;'); + expect(out).toContain(' L12: return sum;'); + // It is the statement list; no inter-symbol section appears without byDepth reach. + expect(out).not.toContain('Inter-procedural symbol reach ('); + // The unified PDG note still surfaces. + expect(out).toContain('Inter-procedural symbol reach is attached'); + }); + + it('renders statement slices with inter-procedural symbol reach when present', () => { + const out = formatImpactResult( + pdgStatementSlice({ + impactedCount: 1, + summary: { direct: 1, processes_affected: 0, modules_affected: 0 }, + byDepthCounts: { 1: 1 }, + byDepth: { + 1: [ + { + depth: 1, + id: 'Function:src/caller.ts:caller', + name: 'caller', + type: 'Function', + filePath: 'src/caller.ts', + processes: [], + }, + ], + }, + pdgInterprocedural: { engine: 'symbol-graph', impactedCount: 1, byDepthCounts: { 1: 1 } }, + }), + ); + + expect(out).toContain('Statements downstream-dependent on src/svc.ts:8 (2):'); + expect(out).toContain('Inter-procedural symbol reach (1):'); + expect(out).toContain('Function caller → src/caller.ts'); + }); + + it('flags slice truncation honestly', () => { + const out = formatImpactResult(pdgStatementSlice({ truncated: true, truncatedBy: 'depth' })); + expect(out).toContain('Truncated'); + expect(out).toContain('by depth'); + }); + + it('flags truncated empty statement slices honestly', () => { + const out = formatImpactResult( + pdgStatementSlice({ + affectedStatements: [], + affectedStatementCount: 0, + truncated: true, + truncatedBy: 'limit', + note: 'Statement slice stopped at the configured result limit.', + }), + ); + + expect(out).toContain('No statements downstream-dependent on src/svc.ts:8'); + expect(out).toContain('Truncated'); + expect(out).toContain('by limit'); + expect(out).toContain('Statement slice stopped at the configured result limit.'); + }); + + it('renders a no-block-at-line result as the steering note, never an empty isolated headline', () => { + // `_runImpactPDG` seedBlocks.length === 0 in statement mode. + const out = formatImpactResult({ + mode: 'pdg', + target: { + id: 'Function:src/svc.ts:accum', + name: 'accum', + type: 'Function', + filePath: 'src/svc.ts', + }, + direction: 'downstream', + criterionLine: 9, + reachableBlocks: [], + blockCount: 0, + affectedStatements: [], + affectedStatementCount: 0, + truncated: false, + depthReached: 0, + epistemic: 'pdg-no-block-at-line', + note: + "No PDG statement block starts at line 9 within 'accum' (src/svc.ts). The line may be " + + "blank, a comment, a brace, or outside the symbol's body. Pass a line that begins an " + + 'executable statement.', + impactedCount: 0, + risk: 'UNKNOWN', + byDepth: {}, + byDepthCounts: { 1: 0 }, + summary: { direct: 0, processes_affected: 0, modules_affected: 0 }, + affected_processes: [], + affected_modules: [], + unresolvedBlockCount: 0, + ambiguousProjectionCount: 0, + }); + expect(out).toContain('No statements downstream-dependent on src/svc.ts:9'); + expect(out).toContain('No PDG statement block starts at line 9'); + expect(out).not.toContain('appears isolated'); + expect(out).not.toContain('PDG-dependent symbols'); + }); + + it('suppresses callgraph next-step hints for PDG and failed impact results', () => { + expect(getNextStepHint('impact')).toContain('Review d=1 items first'); + expect(getNextStepHint('impact', pdgStatementSlice())).toBe(''); + expect(getNextStepHint('impact', { mode: 'pdg', pdgLayer: 'no-layer' })).toBe(''); + expect(getNextStepHint('impact', { mode: 'pdg', status: 'ambiguous' })).toBe(''); + expect(getNextStepHint('impact', { error: 'Target not found' })).toBe(''); + }); +}); + +describe('formatImpactResult — callgraph rendering is UNCHANGED (regression guard)', () => { + // A known callgraph result. The exact rendered string is pinned: U5 must not + // perturb the default-mode output by one byte (KTD8 interchangeability). + const callgraphResult = { + target: { kind: 'Function', name: 'computeTotal' }, + direction: 'upstream', + impactedCount: 2, + risk: 'MEDIUM', + byDepthCounts: { 1: 1, 2: 1 }, + byDepth: { + 1: [ + { + type: 'Function', + name: 'callerA', + filePath: 'src/a.ts', + relationType: 'CALLS', + confidence: 1, + }, + ], + 2: [ + { + type: 'Function', + name: 'callerB', + filePath: 'src/b.ts', + relationType: 'CALLS', + confidence: 0.8, + }, + ], + }, + }; + + it('renders the callgraph result with the exact pre-U5 text (byte-identical)', () => { + const expected = [ + 'Blast radius for Function computeTotal (upstream): 2 symbol(s) depends on this (will break if changed)', + '', + 'd=1: WILL BREAK (direct) (1)', + ' Function callerA → src/a.ts [CALLS]', + '', + 'd=2: LIKELY AFFECTED (indirect) (1)', + ' Function callerB → src/b.ts [CALLS] (conf: 0.8)', + ].join('\n'); + expect(formatImpactResult(callgraphResult)).toBe(expected); + }); + + it('does not apply any PDG framing to a callgraph result', () => { + const out = formatImpactResult(callgraphResult); + expect(out).not.toContain('PDG-dependent symbols'); + expect(out).not.toContain('intra-procedural'); + expect(out).not.toContain('analyze --pdg'); + }); + + it('renders the callgraph summary-only branch unchanged', () => { + const out = formatImpactResult({ + target: { kind: 'Function', name: 'foo' }, + direction: 'downstream', + impactedCount: 3, + risk: 'LOW', + byDepthCounts: { 1: 2, 2: 1 }, + // no byDepth → summary-only branch + }); + expect(out).toContain('(summary only — use summaryOnly: false to see symbol lists)'); + expect(out).toContain('d=1: WILL BREAK (direct) (2)'); + expect(out).not.toContain('PDG-dependent symbols'); + }); + + it('renders the callgraph isolated / zero case unchanged', () => { + const out = formatImpactResult({ + target: { name: 'lonely' }, + direction: 'downstream', + impactedCount: 0, + risk: 'LOW', + }); + expect(out).toBe('lonely: No downstream dependencies found. This symbol appears isolated.'); + }); + + it('renders the callgraph lower-bound (DI/dynamic-dispatch) copy unchanged', () => { + const out = formatImpactResult({ + target: { name: 'viaInterface' }, + direction: 'upstream', + impactedCount: 0, + risk: 'UNKNOWN', + epistemic: 'lower-bound', + boundaries: ['interface PaymentGateway'], + }); + expect(out).toContain('LOWER BOUND'); + expect(out).toContain('interface PaymentGateway'); + expect(out).not.toContain('PDG'); + }); +}); diff --git a/gitnexus/test/unit/eval-server-tool-allowlist.test.ts b/gitnexus/test/unit/eval-server-tool-allowlist.test.ts new file mode 100644 index 000000000..05f05c2df --- /dev/null +++ b/gitnexus/test/unit/eval-server-tool-allowlist.test.ts @@ -0,0 +1,28 @@ +// U8 — the eval-server only dispatches an allowlisted, read-only query surface +// over HTTP. LocalBackend.callTool can also reach write-side / heavier tools +// (rename, shape_check, tool_map, …); the allowlist keeps a stray +// `POST /tool/` from reaching them through the Docker/eval-harness server. + +import { describe, it, expect } from 'vitest'; +import { EVAL_SERVER_TOOLS } from '../../src/cli/eval-server.js'; + +describe('EVAL_SERVER_TOOLS allowlist (U8)', () => { + it('exposes exactly the advertised read-only query surface', () => { + expect([...EVAL_SERVER_TOOLS].sort()).toEqual([ + 'context', + 'cypher', + 'detect_changes', + 'impact', + 'list_repos', + 'query', + ]); + }); + + it('does NOT expose write-side / unadvertised tools', () => { + expect(EVAL_SERVER_TOOLS.has('rename')).toBe(false); + expect(EVAL_SERVER_TOOLS.has('shape_check')).toBe(false); + expect(EVAL_SERVER_TOOLS.has('tool_map')).toBe(false); + expect(EVAL_SERVER_TOOLS.has('group_sync')).toBe(false); + expect(EVAL_SERVER_TOOLS.has('api_impact')).toBe(false); + }); +}); diff --git a/gitnexus/test/unit/group/http-route-graph-method.test.ts b/gitnexus/test/unit/group/http-route-graph-method.test.ts new file mode 100644 index 000000000..0162a3678 --- /dev/null +++ b/gitnexus/test/unit/group/http-route-graph-method.test.ts @@ -0,0 +1,229 @@ +/** + * Step A coverage for issue #2138 groundwork: + * `HttpRouteExtractor.extractProvidersGraph` should read the HTTP verb + * persisted on the Route node (`route.method`, surfaced as `routeMethod` + * by HANDLES_ROUTE_QUERY) as the authoritative method, falling back to + * the edge `reason` only for older indexes / filesystem routes that never + * stored a method. + * + * Why this matters: framework routes (Java Spring, Laravel) are emitted + * with `routeSource = 'framework-route'`, which `methodFromRouteReason` + * cannot decode (returns null). Before the Route node carried `method`, + * the graph path had to re-parse the handler source to recover the verb. + * Persisting the verb on the node removes that dependency for the method + * piece (the handler-name piece is addressed separately in Step B). + * + * Harness mirrors http-route-multi-verb.test.ts: the plugin registry, + * fs-utils, and tree-sitter are mocked so we drive the graph rows + * directly without real grammars. + */ +import { describe, it, expect, vi, beforeEach } from 'vitest'; +import type Parser from 'tree-sitter'; +import type { HttpDetection } from '../../../src/core/group/extractors/http-patterns/types.js'; + +const FILE_DETECTIONS = new Map(); + +vi.mock('../../../src/core/group/extractors/fs-utils.js', () => ({ + readSafe: (_repo: string, _rel: string) => 'stub content', +})); + +vi.mock('../../../src/core/group/extractors/http-patterns/index.js', () => { + return { + HTTP_SCAN_GLOB: '**/*.fake', + getPluginForFile: (rel: string) => ({ + name: 'fake', + language: {}, + scan: (_tree: Parser.Tree) => FILE_DETECTIONS.get(rel) ?? [], + }), + }; +}); + +vi.mock('tree-sitter', () => { + class FakeParser { + setLanguage(_lang: unknown) {} + parse(_src: string) { + return {} as Parser.Tree; + } + } + return { default: FakeParser }; +}); + +import { HttpRouteExtractor } from '../../../src/core/group/extractors/http-route-extractor.js'; + +function detection( + role: 'provider' | 'consumer', + method: string, + p: string, + name: string | null, +): HttpDetection { + return { role, framework: 'test', method, path: p, name, confidence: 0.8 }; +} + +const containsFor = (names: string[]) => + names.map((name) => ({ + uid: `uid-${name}`, + name, + filePath: 'OrderController.java', + labels: ['Method'], + 0: `uid-${name}`, + 1: name, + 2: 'OrderController.java', + 3: ['Method'], + })); + +describe('HttpRouteExtractor — Route.method from graph (Step A / #2138)', () => { + beforeEach(() => { + FILE_DETECTIONS.clear(); + }); + + it('framework-route: uses Route.method when the edge reason cannot decode the verb', async () => { + // Spring controller: reason is the generic 'framework-route', so + // methodFromRouteReason() returns null. The verb must come from the + // Route node's persisted `method` (routeMethod). + FILE_DETECTIONS.set('OrderController.java', [ + detection('provider', 'POST', '/api/orders', 'createOrder'), + ]); + + const db = vi.fn(async (query: string) => { + if (query.includes('HANDLES_ROUTE')) { + return [ + { + fileId: 'f1', + filePath: 'OrderController.java', + routePath: '/api/orders', + routeId: 'r1', + routeMethod: 'POST', + routeSource: 'framework-route', + }, + ]; + } + if (query.includes('CONTAINS')) return containsFor(['createOrder']); + return []; + }); + + const out = await new HttpRouteExtractor().extract(db, '/repo', { + name: 'r', + url: 'r', + } as never); + expect(out).toHaveLength(1); + expect(out[0].meta.method).toBe('POST'); + expect(out[0].contractId).toBe('http::POST::/api/orders'); + }); + + it('framework-route: Route.method disambiguates the handler among multi-verb candidates', async () => { + // Two verbs at the same path in one controller; reason is generic. + // Route.method = PUT must both set the verb AND pick replaceOrder. + FILE_DETECTIONS.set('OrderController.java', [ + detection('provider', 'GET', '/api/orders', 'listOrders'), + detection('provider', 'PUT', '/api/orders', 'replaceOrder'), + ]); + + const db = vi.fn(async (query: string) => { + if (query.includes('HANDLES_ROUTE')) { + return [ + { + fileId: 'f1', + filePath: 'OrderController.java', + routePath: '/api/orders', + routeMethod: 'PUT', + routeSource: 'framework-route', + }, + ]; + } + if (query.includes('CONTAINS')) return containsFor(['listOrders', 'replaceOrder']); + return []; + }); + + const out = await new HttpRouteExtractor().extract(db, '/repo', { + name: 'r', + url: 'r', + } as never); + expect(out).toHaveLength(1); + expect(out[0].meta.method).toBe('PUT'); + expect(out[0].symbolName).toBe('replaceOrder'); + }); + + it('case-insensitive: lower-case Route.method is normalized to an upper-case verb', async () => { + FILE_DETECTIONS.set('OrderController.java', [ + detection('provider', 'DELETE', '/api/orders/{param}', 'deleteOrder'), + ]); + + const db = vi.fn(async (query: string) => { + if (query.includes('HANDLES_ROUTE')) { + return [ + { + fileId: 'f1', + filePath: 'OrderController.java', + routePath: '/api/orders/{id}', + routeMethod: 'delete', + routeSource: 'framework-route', + }, + ]; + } + if (query.includes('CONTAINS')) return containsFor(['deleteOrder']); + return []; + }); + + const out = await new HttpRouteExtractor().extract(db, '/repo', { + name: 'r', + url: 'r', + } as never); + expect(out[0].meta.method).toBe('DELETE'); + }); + + it('backward-compat: missing Route.method falls back to the edge reason (old indexes)', async () => { + // Old index has no `method` on the Route node → routeMethod undefined. + // The decorator reason still decodes the verb as before. + FILE_DETECTIONS.set('routes.ts', [detection('provider', 'GET', '/api/orders', 'listOrders')]); + + const db = vi.fn(async (query: string) => { + if (query.includes('HANDLES_ROUTE')) { + return [ + { + fileId: 'f1', + filePath: 'routes.ts', + routePath: '/api/orders', + // no routeMethod field at all + routeSource: 'decorator-Get', + }, + ]; + } + if (query.includes('CONTAINS')) return containsFor(['listOrders']); + return []; + }); + + const out = await new HttpRouteExtractor().extract(db, '/repo', { + name: 'r', + url: 'r', + } as never); + expect(out[0].meta.method).toBe('GET'); + expect(out[0].symbolName).toBe('listOrders'); + }); + + it('backward-compat: no Route.method and undecodable reason stays at conservative GET', async () => { + FILE_DETECTIONS.set('routes.ts', [detection('provider', 'POST', '/api/orders', 'createOrder')]); + + const db = vi.fn(async (query: string) => { + if (query.includes('HANDLES_ROUTE')) { + return [ + { + fileId: 'f1', + filePath: 'routes.ts', + routePath: '/api/orders', + routeSource: 'framework-route', // undecodable, and no routeMethod + }, + ]; + } + if (query.includes('CONTAINS')) return containsFor(['createOrder']); + return []; + }); + + const out = await new HttpRouteExtractor().extract(db, '/repo', { + name: 'r', + url: 'r', + } as never); + // Single candidate, so its method is adopted (existing behavior); the + // point is that absence of routeMethod does not throw and still works. + expect(out[0].meta.method).toBe('POST'); + }); +}); diff --git a/gitnexus/test/unit/impact-pdg-ascent-note.test.ts b/gitnexus/test/unit/impact-pdg-ascent-note.test.ts new file mode 100644 index 000000000..c18cff61a --- /dev/null +++ b/gitnexus/test/unit/impact-pdg-ascent-note.test.ts @@ -0,0 +1,89 @@ +// U7 — the impact result note tells a non-TypeScript/JavaScript user that +// return-value ascent (FU-C) is currently TS/JS-only, instead of silently +// suppressing the guidance because the CALL_SUMMARY layer flag is set. Only the +// TS/JS harvester records the formal-index ascent needs, so for any other +// language the ascent is structurally empty. + +import { describe, expect, it } from 'vitest'; +import { runImpactPDG, type RunPdgImpactDeps } from '../../src/mcp/local/pdg-impact.js'; + +// A mock that drives ONE real inter-procedural descent hop: the criterion's +// reachable block calls `helper`, the descent resolves helper's span (so +// interproceduralHops > 0 and the note block fires). The criterion file's +// extension selects the language the U7 note keys on. +function descentExec(file: string): RunPdgImpactDeps['executeParameterized'] { + const seed = `BasicBlock:${file}:1:0:0`; + const callBlock = `BasicBlock:${file}:1:0:2`; + const calleeSeed = `BasicBlock:${file}:5:0:0`; + let bfs = 0; + return async (_repo, query) => { + // Top-level seed fetch is line-anchored (`a.startLine = $line`); the descent's + // callee seed fetch is range-anchored — route by that. + if (query.includes('RETURN a.id AS id LIMIT')) { + return query.includes('a.startLine = $line') ? [{ id: seed }] : [{ id: calleeSeed }]; + } + if (query.includes('MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock)')) { + bfs += 1; + return bfs === 1 ? [{ id: callBlock }] : []; + } + if (query.includes('RETURN b.id AS id, b.calleeIds AS calleeIds')) { + return [{ id: callBlock, calleeIds: `Function:${file}:helper` }]; + } + if (query.includes("r.type = 'CALL_SUMMARY'")) return []; + if (query.includes('s.id IN $ids') && query.includes('AS filePath')) { + return [{ id: `Function:${file}:helper`, filePath: file, startLine: 4, endLine: 6 }]; + } + if (query.includes('MATCH (b:BasicBlock) WHERE b.id IN $ids')) { + return [ + { id: seed, line: 1, endLine: 1, text: 'run()' }, + { id: callBlock, line: 3, endLine: 3, text: 'x = helper()' }, + { id: calleeSeed, line: 5, endLine: 5, text: 'return 1' }, + ]; + } + if (query.includes('MATCH (s:`Function`)')) return []; + return []; + }; +} + +const run = (file: string, callSummaryAvailable: boolean) => + runImpactPDG({ + repo: { lbugPath: 'repo' }, + sym: { id: `Function:${file}:run`, name: 'run', filePath: file, startLine: 0, endLine: 7 }, + symType: 'Function', + direction: 'downstream', + maxDepth: 3, + limit: 50, + line: 1, + executeParameterized: descentExec(file), + callSummaryAvailable, + }); + +const CAVEAT = 'return-value ascent is currently TypeScript/JavaScript-only'; + +describe('runImpactPDG — TS/JS-only ascent note (U7)', () => { + it('non-TS/JS (.py) criterion with CALL_SUMMARY present → notes ascent is TS/JS-only', async () => { + const result = await run('src/svc.py', true); + expect('affectedStatements' in result).toBe(true); + const note = 'affectedStatements' in result ? (result.note ?? '') : ''; + expect(note).toContain(CAVEAT); + }); + + it('TypeScript (.ts) criterion → no TS/JS-only caveat (ascent applies)', async () => { + const result = await run('src/svc.ts', true); + const note = 'affectedStatements' in result ? (result.note ?? '') : ''; + expect(note).not.toContain(CAVEAT); + }); + + it('JavaScript (.js) criterion → no TS/JS-only caveat (JS also sets formalIndex)', async () => { + const result = await run('src/svc.js', true); + const note = 'affectedStatements' in result ? (result.note ?? '') : ''; + expect(note).not.toContain(CAVEAT); + }); + + it('v3 index (callSummaryAvailable false) → re-index note, not the language caveat', async () => { + const result = await run('src/svc.py', false); + const note = 'affectedStatements' in result ? (result.note ?? '') : ''; + expect(note).toContain('re-index for CALL_SUMMARY'); + expect(note).not.toContain(CAVEAT); + }); +}); diff --git a/gitnexus/test/unit/impact-pdg-blast-radius-metrics.test.ts b/gitnexus/test/unit/impact-pdg-blast-radius-metrics.test.ts new file mode 100644 index 000000000..75f8efc51 --- /dev/null +++ b/gitnexus/test/unit/impact-pdg-blast-radius-metrics.test.ts @@ -0,0 +1,117 @@ +import { describe, expect, it } from 'vitest'; + +import { + median, + parseMarkdownRows, + summarizeBlastRadius, + symbolSetFromByDepth, +} from '../../bench/impact-pdg/blast-radius.mjs'; + +describe('impact-pdg blast-radius metric helpers', () => { + it('parses a cypher markdown table into row objects', () => { + const md = [ + '| id | startLine |', + '| --- | --- |', + '| BasicBlock:a.ts:5:2:0 | 6 |', + '| BasicBlock:a.ts:5:2:1 | 9 |', + ].join('\n'); + + expect(parseMarkdownRows(md)).toEqual([ + { id: 'BasicBlock:a.ts:5:2:0', startLine: '6' }, + { id: 'BasicBlock:a.ts:5:2:1', startLine: '9' }, + ]); + expect(parseMarkdownRows('')).toEqual([]); + expect(parseMarkdownRows('| n |\n| --- |')).toEqual([]); + }); + + it('builds a stable symbol-id set from a byDepth record', () => { + const set = symbolSetFromByDepth({ + 1: [ + { id: 'Function:src/a.ts:a', name: 'a', filePath: 'src/a.ts' }, + { id: '', name: 'dynamic', filePath: 'src/b.ts' }, + ], + 2: [{ id: 'Method:src/c.ts:C.m', name: 'm', filePath: 'src/c.ts' }], + }); + + expect([...set].sort()).toEqual([ + 'Function:src/a.ts:a', + 'Method:src/c.ts:C.m', + 'dynamic@src/b.ts', + ]); + }); + + it('summarizes localization, inter-symbol agreement, and latency', () => { + const summary = summarizeBlastRadius([ + { + bodyBlocks: 20, + sliceBlocks: 4, + ratio: 0.2, + callgraphSymbols: 6, + statementPreciseSymbols: 3, + statementPrecision: 0.5, + pdgOnly: 0, + cgOnly: 0, + callgraphMs: 100, + pdgMs: 150, + }, + { + bodyBlocks: 16, + sliceBlocks: 8, + ratio: 0.5, + callgraphSymbols: 4, + statementPreciseSymbols: 4, + statementPrecision: 1, + pdgOnly: 0, + cgOnly: 0, + callgraphMs: 80, + pdgMs: 120, + }, + { + bodyBlocks: 10, + sliceBlocks: 10, + ratio: 1, + callgraphSymbols: 3, + statementPreciseSymbols: 1, + statementPrecision: 0.333, + pdgOnly: 2, + cgOnly: 1, + callgraphMs: 60, + pdgMs: 90, + }, + ]); + + expect(summary.n).toBe(3); + expect(summary.localization).toMatchObject({ + medianSliceOverBody: 0.5, + meanSliceOverBody: 0.567, + medianBodyBlocks: 16, + medianSliceBlocks: 8, + casesSliceSmallerThanBody: 2, + }); + expect(summary.interSymbol).toMatchObject({ + casesPdgFindsMore: 1, + casesPdgFindsFewer: 1, + casesIdentical: 2, + totalPdgOnlySymbols: 2, + totalCgOnlySymbols: 1, + }); + expect(summary.latency).toMatchObject({ + medianCallgraphMs: 80, + medianPdgMs: 120, + medianPdgOverCallgraph: 1.5, + }); + expect(summary.statementPrecise).toMatchObject({ + casesWithSlice: 3, + casesTighterThanCallgraph: 2, // 3<6 and 1<3; the 4==4 case does not narrow + medianStatementPrecision: 0.5, + medianPreciseSymbols: 3, + medianCallgraphSymbols: 4, + }); + }); + + it('computes median deterministically (odd and even lengths)', () => { + expect(median([3, 1, 2])).toBe(2); + expect(median([4, 1, 2, 3])).toBe(2.5); + expect(median([])).toBe(null); + }); +}); diff --git a/gitnexus/test/unit/impact-pdg-compose-dedup.test.ts b/gitnexus/test/unit/impact-pdg-compose-dedup.test.ts new file mode 100644 index 000000000..7b4f0c010 --- /dev/null +++ b/gitnexus/test/unit/impact-pdg-compose-dedup.test.ts @@ -0,0 +1,132 @@ +// U5 regression — composeUnifiedPdgImpactResult must NOT double-count a symbol +// reached by BOTH the local PDG block-expansion AND the inter-procedural +// callgraph layer. The headline impactedCount / summary.direct are the distinct +// union across layers; byDepthCounts is per-depth distinct (so a symbol reached +// at two different depths legitimately appears in two depth buckets, and +// sum(byDepthCounts) can exceed impactedCount — documented, not a bug). risk +// must stay UNKNOWN (never lowered by the new count). + +import { describe, it, expect } from 'vitest'; +import { + composeUnifiedPdgImpactResult, + type PdgImpactSuccessResult, +} from '../../src/mcp/local/pdg-impact.js'; + +interface ByDepthItem { + depth: number; + id: string | null; + name: string; + type: string; + filePath: string; +} + +const item = (id: string | null, depth: number): ByDepthItem => ({ + depth, + id, + name: id ?? '(unresolved)', + type: 'Function', + filePath: 'src/a.ts', +}); + +const countsOf = (byDepth: Record): Record => + Object.fromEntries(Object.entries(byDepth).map(([d, items]) => [Number(d), items.length])); + +const local = ( + byDepth: Record, + impactedCount: number, +): PdgImpactSuccessResult => ({ + mode: 'pdg', + pdgResultVersion: 1, + target: { id: 'T', name: 'criterion', type: 'Function', filePath: 'src/a.ts' }, + direction: 'downstream', + risk: 'UNKNOWN', + epistemic: 'pdg-intra-procedural', + impactedCount, + byDepth, + byDepthCounts: countsOf(byDepth), + summary: { direct: impactedCount, processes_affected: 0, modules_affected: 0 }, + affected_processes: [], + affected_modules: [], + partial: false, + reachableBlocks: [], + intraReachableBlocks: [], + seedBlocks: [], + blockCount: 0, + affectedStatements: [], + affectedStatementCount: 0, + depthReached: 1, + unresolvedBlockCount: 0, + ambiguousProjectionCount: 0, +}); + +const interproc = ( + byDepth: Record, + impactedCount: number, + direct: number, +) => ({ + byDepth, + byDepthCounts: countsOf(byDepth), + impactedCount, + summary: { direct, processes_affected: 0, modules_affected: 0 }, + affected_processes: [], + affected_modules: [], + partial: false, +}); + +describe('composeUnifiedPdgImpactResult — cross-bucket dedup (U5)', () => { + it('same-depth overlap → counted once (was local+interproc sum)', () => { + // local {A,B}@d1, interproc {A,B}@d1 — both layers reach the same two symbols. + const result = composeUnifiedPdgImpactResult( + local({ 1: [item('A', 1), item('B', 1)] }, 2), + interproc({ 1: [item('B', 1), item('A', 1)] }, 2, 2), + ); + // distinct {A,B} = 2 (NOT 2+2 = 4). + expect(result).toMatchObject({ + impactedCount: 2, + risk: 'UNKNOWN', + byDepthCounts: { 1: 2 }, + summary: { direct: 2 }, + }); + }); + + it('disjoint layers → dedup is a no-op (full sum)', () => { + const result = composeUnifiedPdgImpactResult( + local({ 1: [item('A', 1)] }, 1), + interproc({ 1: [item('B', 1)] }, 1, 1), + ); + expect(result).toMatchObject({ + impactedCount: 2, + byDepthCounts: { 1: 2 }, + summary: { direct: 2 }, + }); + }); + + it('cross-depth same symbol → counted once in headline, retained per-depth', () => { + // C is local@d1 AND interproc@d2 — one distinct symbol, two legitimate buckets. + const result = composeUnifiedPdgImpactResult( + local({ 1: [item('C', 1)] }, 1), + interproc({ 2: [item('C', 2)] }, 1, 0), + ); + expect(result).toMatchObject({ + impactedCount: 1, // distinct + byDepthCounts: { 1: 1, 2: 1 }, // sum 2 > impactedCount 1, documented + summary: { direct: 1 }, + }); + }); + + it('unresolved (null-id) item is excluded from the distinct count', () => { + const result = composeUnifiedPdgImpactResult( + local({ 1: [item('A', 1), item(null, 1)] }, 1), + interproc({ 1: [item('A', 1)] }, 1, 1), + ); + expect(result).toMatchObject({ impactedCount: 1, risk: 'UNKNOWN' }); + }); + + it('no inter-procedural layer → local count unchanged', () => { + const result = composeUnifiedPdgImpactResult( + local({ 1: [item('A', 1), item('B', 1)] }, 2), + null, + ); + expect(result).toMatchObject({ impactedCount: 2, summary: { direct: 2 } }); + }); +}); diff --git a/gitnexus/test/unit/impact-pdg-id-bridge-gate.test.ts b/gitnexus/test/unit/impact-pdg-id-bridge-gate.test.ts new file mode 100644 index 000000000..6c08ec43d --- /dev/null +++ b/gitnexus/test/unit/impact-pdg-id-bridge-gate.test.ts @@ -0,0 +1,143 @@ +// U9 — resolved-symbol-id soundness GATE scorer for the impact-PDG accuracy +// harness (plan 2026-06-18-001 U9; Covers R1, R5). +// +// Asserts the PURE gate helpers (`idProvenIdsFromResult`, `evaluateIdBridge`) +// that `bench/impact-pdg/measure.mjs` uses to gate the `intra-overloaded-callee` +// fixture — on SYNTHETIC impact-result shapes ONLY, no LocalBackend / analyze / +// DB. The helpers live in the build-free `.mjs` harness, imported directly here, +// so this test is deterministic and stays OUT of the flaky full-pipeline lane +// (mirroring impact-pdg-id-vs-name-metrics.test.ts). The live id-bridge axis +// runs only via `node --import tsx bench/impact-pdg/measure.mjs`, never in +// `npm test`. + +import { describe, expect, it } from 'vitest'; +// @ts-expect-error — .mjs pure-JS harness module, no types (intentional; build-free). +import * as M from '../../bench/impact-pdg/measure.mjs'; + +interface ReachedItem { + readonly id?: string; + readonly name?: string; + readonly filePath?: string; + readonly pdgEvidence?: string; +} + +interface IdBridgeExpectation { + readonly seedLine?: number; + readonly idProven?: readonly string[]; + readonly nameWouldProve?: readonly string[]; + readonly fpEliminated?: readonly string[]; +} + +interface IdBridgeVerdict { + readonly ok: boolean; + readonly problems: readonly string[]; + readonly idProven: readonly string[]; + readonly nameProven: readonly string[]; + readonly fpEliminated: readonly string[]; + readonly over: number; +} + +interface PdgInterproceduralShape { + readonly statementPreciseByDepth?: Record; +} +interface PdgResultShape { + readonly pdgInterprocedural?: PdgInterproceduralShape; +} + +const idProvenIdsFromResult = M.idProvenIdsFromResult as (res: PdgResultShape) => string[]; +const evaluateIdBridge = M.evaluateIdBridge as ( + idProven: readonly string[], + nameWouldProve: readonly string[], + expected: IdBridgeExpectation, +) => IdBridgeVerdict; + +// The fixture's two collision callees — same leaf `process`, distinct ids. +const ALPHA = 'Method:src/route.ts:Alpha.process#1'; +const BETA = 'Method:src/route.ts:Beta.process#1'; + +const item = (id: string): ReachedItem => ({ + id, + name: 'process', + filePath: 'src/route.ts', + pdgEvidence: 'callgraph-bridge', +}); + +describe('impact-pdg id-bridge gate — idProvenIdsFromResult()', () => { + it('flattens statementPreciseByDepth across depths into a sorted id set', () => { + const res: PdgResultShape = { + pdgInterprocedural: { + statementPreciseByDepth: { 1: [item(ALPHA)], 2: [item(BETA)] }, + }, + }; + expect(idProvenIdsFromResult(res)).toEqual([ALPHA, BETA].sort()); + }); + + it('returns the single proven id when only the on-slice callee is statement-precise', () => { + // The U9 soundness case: Beta is reached but unproven-bridge, so it is NOT in + // statementPreciseByDepth — only Alpha (the on-slice callee) appears. + const res: PdgResultShape = { + pdgInterprocedural: { statementPreciseByDepth: { 1: [item(ALPHA)] } }, + }; + expect(idProvenIdsFromResult(res)).toEqual([ALPHA]); + }); + + it('falls back to name@filePath for an id-less reached item, never dropping it', () => { + const res: PdgResultShape = { + pdgInterprocedural: { + statementPreciseByDepth: { 1: [{ name: 'dyn', filePath: 'd.ts' }] }, + }, + }; + expect(idProvenIdsFromResult(res)).toEqual(['dyn@d.ts']); + }); + + it('returns an empty set when there is no statement-precise reach', () => { + expect(idProvenIdsFromResult({})).toEqual([]); + }); +}); + +describe('impact-pdg id-bridge gate — evaluateIdBridge()', () => { + const expected: IdBridgeExpectation = { + seedLine: 32, + idProven: [ALPHA], + nameWouldProve: [ALPHA, BETA], + fpEliminated: [BETA], + }; + + it('PASSES the soundness gate: id proves exactly Alpha, name over-attributes Beta', () => { + const v = evaluateIdBridge([ALPHA], [ALPHA, BETA], expected); + expect(v).toMatchObject({ ok: true, over: 1 }); + expect(v.problems).toEqual([]); + expect(v.idProven).toEqual([ALPHA]); + expect(v.fpEliminated).toEqual([BETA]); + }); + + it('FAILS when the id-proven set over-attributes (proves both — the name-bridge bug)', () => { + // If the id bridge regressed to prove BOTH, the id set != ground-truth single + // id AND the name match no longer over-attributes — the gate must fail. + const v = evaluateIdBridge([ALPHA, BETA], [ALPHA, BETA], expected); + expect(v).toMatchObject({ ok: false, over: 0 }); + expect(v.problems.length).toBeGreaterThan(0); + }); + + it('FAILS when the id-proven set drops the correct callee (under-attribution)', () => { + const v = evaluateIdBridge([], [ALPHA, BETA], expected); + expect(v).toMatchObject({ ok: false }); + expect(v.problems.join(' ')).toContain('id-proven set'); + }); + + it('FAILS when name-match does not over-attribute the expected collision id', () => { + // id-proven matches, but the name counterfactual proves only Alpha (the slice + // lost its collision power) — fpEliminated would be empty → loud failure. + const v = evaluateIdBridge([ALPHA], [ALPHA], expected); + expect(v).toMatchObject({ ok: false, over: 0 }); + expect(v.problems.join(' ')).toContain('over-attribute'); + }); + + it('is order- and duplicate-independent over its inputs (determinism)', () => { + const a = evaluateIdBridge([ALPHA], [BETA, ALPHA, BETA], expected); + const b = evaluateIdBridge([ALPHA, ALPHA], [ALPHA, BETA], expected); + expect(a).toMatchObject({ ok: true, over: 1 }); + expect(b).toMatchObject({ ok: true, over: 1 }); + expect(a.fpEliminated).toEqual(b.fpEliminated); + }); +}); diff --git a/gitnexus/test/unit/impact-pdg-id-vs-name-metrics.test.ts b/gitnexus/test/unit/impact-pdg-id-vs-name-metrics.test.ts new file mode 100644 index 000000000..757ee8668 --- /dev/null +++ b/gitnexus/test/unit/impact-pdg-id-vs-name-metrics.test.ts @@ -0,0 +1,265 @@ +// U8 — id-vs-name metric scorer for the realized PDG name-collision proof harness. +// +// Asserts the PURE scorer (`scoreIdVsName` / `summarizeIdVsName` / +// `reachedItemKey`) on SYNTHETIC reached-item sets ONLY — no `LocalBackend`, no +// analyze, no DB. The scorer lives in `bench/impact-pdg/name-collision.mjs`, +// imported here directly, so this test is deterministic and stays OUT of the +// flaky full-pipeline lane (mirroring impact-pdg-metric-math.test.ts and +// impact-pdg-blast-radius-metrics.test.ts). The live substrate runs only via +// `node --import tsx bench/impact-pdg/name-collision.mjs`, never in `npm test`. + +import { describe, expect, it } from 'vitest'; +// @ts-expect-error — .mjs pure-JS module, no types; intentional (build-free harness). +import * as M from '../../bench/impact-pdg/name-collision.mjs'; + +interface ReachedItem { + readonly id?: string; + readonly name?: string; + readonly filePath?: string; +} + +interface IdVsNameScore { + readonly nameProven: number; + readonly idProven: number; + readonly fpEliminated: number; + readonly fnRecovered: number; + readonly fpEliminatedKeys: readonly string[]; + readonly fnRecoveredKeys: readonly string[]; +} + +interface IdVsNameSummary { + readonly n: number; + readonly totalNameProven: number; + readonly totalIdProven: number; + readonly totalFpEliminated: number; + readonly totalFnRecovered: number; + readonly functionsWithFpEliminated: number; + readonly functionsWithFnRecovered: number; + readonly fpEliminatedRate: number | null; + readonly fnRecoveredRate: number | null; +} + +interface ProvenSets { + readonly nameProven: readonly ReachedItem[]; + readonly idProven: readonly ReachedItem[]; + readonly discriminating: boolean; + readonly wholeSymbol: boolean; + readonly truncated: boolean; +} + +const reachedItemKey = M.reachedItemKey as (item: ReachedItem) => string; +const scoreIdVsName = M.scoreIdVsName as ( + nameProvenItems: readonly ReachedItem[], + idProvenItems: readonly ReachedItem[], +) => IdVsNameScore; +const summarizeIdVsName = M.summarizeIdVsName as ( + cases: ReadonlyArray & { discriminatingSlice?: boolean }>, +) => IdVsNameSummary; +const bridgeProvenSets = M.bridgeProvenSets as ( + reachedItems: readonly ReachedItem[], + sliceCalleeNames: ReadonlySet, + sliceCalleeIds: ReadonlySet, +) => ProvenSets; + +const names = (...ns: string[]): ReadonlySet => new Set(ns); +const ids = (...xs: string[]): ReadonlySet => new Set(xs); + +interface KeyedItem { + readonly id: string; + readonly name: string; + readonly filePath: string; +} + +// Two distinct overloads sharing the leaf name `serialize` — the collision shape: +// same name, different resolved ids. `id` is required so no narrowing cast is +// needed when these fixtures seed the slice id sets. +const SER1: KeyedItem = { + id: 'Method:Ser.java:S.serialize#1', + name: 'serialize', + filePath: 'Ser.java', +}; +const SER2: KeyedItem = { + id: 'Method:Ser.java:S.serialize#2', + name: 'serialize', + filePath: 'Ser.java', +}; +// An import-alias callee: proven by id even though its leaf name is absent from +// the slice's `callees` (the false-negative the name match would miss). +const ALIAS: KeyedItem = { id: 'Function:a.ts:realFn', name: 'aliased', filePath: 'a.ts' }; +const FOO: KeyedItem = { id: 'Function:a.ts:foo', name: 'foo', filePath: 'a.ts' }; + +describe('impact-pdg id-vs-name scorer — reachedItemKey()', () => { + it('keys by resolved id when present', () => { + expect(reachedItemKey(SER1)).toBe('Method:Ser.java:S.serialize#1'); + expect(reachedItemKey(SER2)).toBe('Method:Ser.java:S.serialize#2'); + }); + + it('falls back to name@filePath only when id is absent', () => { + expect(reachedItemKey({ name: 'dyn', filePath: 'd.ts' })).toBe('dyn@d.ts'); + expect(reachedItemKey({ name: 'dyn' })).toBe('dyn@(unknown)'); + expect(reachedItemKey({})).toBe('(unknown)@(unknown)'); + }); +}); + +describe('impact-pdg id-vs-name scorer — bridgeProvenSets() predicate replica', () => { + it('discriminating collision: name proves both overloads, id proves only the on-slice one', () => { + // slice `callees` = {serialize}; slice `calleeIds` = {serialize#2}. The name + // bridge proves BOTH SER1 and SER2 (leaf `serialize` on slice); the id bridge + // proves only SER2 (its resolved id is on the slice). SER1 = collision FP. + const sets = bridgeProvenSets([SER1, SER2], names('serialize'), ids(SER2.id)); + expect(sets).toMatchObject({ discriminating: true, wholeSymbol: false, truncated: false }); + expect(sets.nameProven.map(reachedItemKey).sort()).toEqual([SER1.id, SER2.id].sort()); + expect(sets.idProven.map(reachedItemKey)).toEqual([SER2.id]); + }); + + it('discriminating alias: id proves the alias whose leaf name is absent from the slice callees', () => { + // slice `callees` = {foo} (NOT `aliased`); slice `calleeIds` = {foo, realFn}. + // The name bridge proves only FOO; the id bridge also proves ALIAS via its id. + const sets = bridgeProvenSets([FOO, ALIAS], names('foo'), ids(FOO.id, ALIAS.id)); + expect(sets.nameProven.map(reachedItemKey)).toEqual([FOO.id]); + expect(sets.idProven.map(reachedItemKey).sort()).toEqual([FOO.id, ALIAS.id].sort()); + }); + + it('whole-symbol fallback (empty slice callees): BOTH bridges prove ALL reached items', () => { + // The bug the earlier draft hit: an empty-callees seed block must NOT read as + // "name proves nothing" — the bridge whole-symbol-falls-back and proves all. + const sets = bridgeProvenSets([SER1, SER2], names(), ids(SER2.id)); + expect(sets).toMatchObject({ wholeSymbol: true, discriminating: false }); + expect(sets.nameProven.map(reachedItemKey).sort()).toEqual([SER1.id, SER2.id].sort()); + expect(sets.idProven.map(reachedItemKey).sort()).toEqual([SER1.id, SER2.id].sort()); + }); + + it('sentinel fallback (capped block): BOTH bridges prove ALL (callee-unknown)', () => { + // `*` in the slice names marks an incomplete (capped) callee list → callgraph-equal. + const sets = bridgeProvenSets([SER1, SER2], names('serialize', '*'), ids(SER2.id)); + expect(sets).toMatchObject({ truncated: true, discriminating: false }); + expect(sets.nameProven.map(reachedItemKey).sort()).toEqual([SER1.id, SER2.id].sort()); + expect(sets.idProven.map(reachedItemKey).sort()).toEqual([SER1.id, SER2.id].sort()); + }); + + it('pre-v3 (no calleeIds): id set degrades to the name path ⇒ identical to name set', () => { + const sets = bridgeProvenSets([SER1, SER2], names('serialize'), ids()); + expect(sets).toMatchObject({ discriminating: false, wholeSymbol: false, truncated: false }); + expect(sets.idProven.map(reachedItemKey).sort()).toEqual( + sets.nameProven.map(reachedItemKey).sort(), + ); + }); + + it('is deterministic: same inputs ⇒ identical proven sets', () => { + const a = bridgeProvenSets([SER1, SER2, FOO], names('serialize', 'foo'), ids(SER2.id, FOO.id)); + const b = bridgeProvenSets([SER1, SER2, FOO], names('serialize', 'foo'), ids(SER2.id, FOO.id)); + expect(a.nameProven).toEqual(b.nameProven); + expect(a.idProven).toEqual(b.idProven); + }); +}); + +describe('impact-pdg id-vs-name scorer — scoreIdVsName()', () => { + it('name-set ⊋ id-set (a collision): fpEliminated counts exactly the extra name-only labels', () => { + // Two same-named overloads are BOTH name-proven (the leaf `serialize` is on the + // slice), but only SER2 is id-proven (the slice resolves to #2). The id bridge + // drops SER1 — one realized collision false-positive eliminated. + const score = scoreIdVsName([SER1, SER2], [SER2]); + expect(score).toMatchObject({ + nameProven: 2, + idProven: 1, + fpEliminated: 1, + fnRecovered: 0, + fpEliminatedKeys: ['Method:Ser.java:S.serialize#1'], + fnRecoveredKeys: [], + }); + }); + + it('id-set has an alias-only member: fnRecovered counts it (import-alias FN recovered)', () => { + // ALIAS is id-proven but NOT name-proven (its leaf name is absent from the + // slice `callees`) — the name bridge would miss it; the id bridge recovers it. + const score = scoreIdVsName([FOO], [FOO, ALIAS]); + expect(score).toMatchObject({ + nameProven: 1, + idProven: 2, + fpEliminated: 0, + fnRecovered: 1, + fpEliminatedKeys: [], + fnRecoveredKeys: ['Function:a.ts:realFn'], + }); + }); + + it('identical sets ⇒ fpEliminated == 0 and fnRecovered == 0', () => { + const score = scoreIdVsName([SER2, FOO], [FOO, SER2]); + expect(score).toMatchObject({ + nameProven: 2, + idProven: 2, + fpEliminated: 0, + fnRecovered: 0, + fpEliminatedKeys: [], + fnRecoveredKeys: [], + }); + }); + + it('is deterministic: same input ⇒ identical output regardless of item ordering', () => { + const a = scoreIdVsName([SER1, SER2, FOO], [SER2, ALIAS]); + const b = scoreIdVsName([FOO, SER2, SER1], [ALIAS, SER2]); + // Order-independent and stable (keys sorted, no Date/random). + expect(a).toEqual(b); + expect(a).toMatchObject({ + fpEliminated: 2, // SER1 + FOO are name-proven but not id-proven + fnRecovered: 1, // ALIAS is id-proven but not name-proven + fpEliminatedKeys: ['Function:a.ts:foo', 'Method:Ser.java:S.serialize#1'], + fnRecoveredKeys: ['Function:a.ts:realFn'], + }); + }); + + it('id-less reached items diff by name@filePath fallback key', () => { + const dynA: ReachedItem = { name: 'handler', filePath: 'x.ts' }; + const dynB: ReachedItem = { name: 'handler', filePath: 'y.ts' }; + // Same leaf name, different files ⇒ distinct keys ⇒ a real diff. + const score = scoreIdVsName([dynA, dynB], [dynA]); + expect(score).toMatchObject({ + nameProven: 2, + idProven: 1, + fpEliminated: 1, + fnRecovered: 0, + fpEliminatedKeys: ['handler@y.ts'], + fnRecoveredKeys: [], + }); + }); +}); + +describe('impact-pdg id-vs-name scorer — summarizeIdVsName()', () => { + it('aggregates per-function counts and rates over the case set', () => { + const summary = summarizeIdVsName([ + // function 1: a collision eliminated, no alias recovered + { nameProven: 2, idProven: 1, fpEliminated: 1, fnRecovered: 0 }, + // function 2: an alias recovered, no collision + { nameProven: 1, idProven: 2, fpEliminated: 0, fnRecovered: 1 }, + // function 3: name == id (no change) + { nameProven: 3, idProven: 3, fpEliminated: 0, fnRecovered: 0 }, + ]); + expect(summary).toMatchObject({ + n: 3, + totalNameProven: 6, + totalIdProven: 6, + totalFpEliminated: 1, + totalFnRecovered: 1, + functionsWithFpEliminated: 1, + functionsWithFnRecovered: 1, + // round(1/6, 3) — the harness rounds rates to 3 digits (see blast-radius.mjs). + fpEliminatedRate: 0.167, + fnRecoveredRate: 0.167, + }); + }); + + it('returns null rates when there are no proven labels (no division by zero)', () => { + const summary = summarizeIdVsName([]); + expect(summary).toMatchObject({ + n: 0, + totalNameProven: 0, + totalIdProven: 0, + totalFpEliminated: 0, + totalFnRecovered: 0, + functionsWithFpEliminated: 0, + functionsWithFnRecovered: 0, + fpEliminatedRate: null, + fnRecoveredRate: null, + }); + }); +}); diff --git a/gitnexus/test/unit/impact-pdg-metric-math.test.ts b/gitnexus/test/unit/impact-pdg-metric-math.test.ts new file mode 100644 index 000000000..e52051254 --- /dev/null +++ b/gitnexus/test/unit/impact-pdg-metric-math.test.ts @@ -0,0 +1,617 @@ +// U7 — metric-math unit test for the impact-PDG accuracy scorer. +// +// Asserts the scorer arithmetic (precision / recall / F1 / Jaccard / set-diffs / +// aggregation / annotation fingerprint) on SYNTHETIC CIS/AIS sets ONLY — no +// `runPipelineFromRepo`, no `analyze`, no `LocalBackend`, no DB. The pure +// scorer lives in `bench/impact-pdg/metrics.mjs`, imported here directly, so +// this test is deterministic and stays OUT of the flaky full-pipeline lane +// (Arch-review Issue 5). The live substrate is exercised manually by +// `measure.mjs`, never in `npm test`. + +import { describe, it, expect } from 'vitest'; +// @ts-expect-error — .mjs pure-JS module, no types; intentional (build-free harness). +import * as M from '../../bench/impact-pdg/metrics.mjs'; + +const k = (sym: string, file = 'src/a.ts') => M.symbolKey(sym, file); +const setOf = (...syms: string[]) => M.toKeySet(syms.map((s) => k(s))); + +describe('impact-pdg metric math — score()', () => { + it('computes precision/recall/F1 on a known partial overlap', () => { + // CIS = {a,b,c}, AIS = {b,c,d}. TP = {b,c} = 2. + const cis = setOf('a', 'b', 'c'); + const ais = setOf('b', 'c', 'd'); + const s = M.score(cis, ais); + expect(s.tp).toBe(2); + expect(s.precision).toBeCloseTo(2 / 3, 12); // 2 of 3 predicted are real + expect(s.recall).toBeCloseTo(2 / 3, 12); // 2 of 3 real are found + expect(s.f1).toBeCloseTo(2 / 3, 12); // p==r ⇒ F1==p + expect(s.fpis).toEqual([k('a')]); // CIS−AIS + expect(s.fnis).toEqual([k('d')]); // AIS−CIS + expect(s.fpisCount).toBe(1); + expect(s.fnisCount).toBe(1); + expect(s.cisAisRatio).toBeCloseTo(1, 12); + }); + + it('perfect match ⇒ P=R=F1=1, empty diffs', () => { + const s = M.score(setOf('a', 'b'), setOf('a', 'b')); + expect(s.precision).toBe(1); + expect(s.recall).toBe(1); + expect(s.f1).toBe(1); + expect(s.fpis).toEqual([]); + expect(s.fnis).toEqual([]); + }); + + it('asymmetric F1: high recall, low precision', () => { + // CIS over-approximates: {a,b,c,d}, AIS = {a}. TP=1. + const s = M.score(setOf('a', 'b', 'c', 'd'), setOf('a')); + expect(s.precision).toBeCloseTo(1 / 4, 12); + expect(s.recall).toBe(1); + // F1 = 2*(0.25*1)/(0.25+1) = 0.5/1.25 = 0.4 + expect(s.f1).toBeCloseTo(0.4, 12); + expect(s.cisAisRatio).toBeCloseTo(4, 12); // 4× over-approx + expect(s.fpisCount).toBe(3); + expect(s.fnisCount).toBe(0); + }); + + it('disjoint sets ⇒ P=R=F1=0', () => { + const s = M.score(setOf('a', 'b'), setOf('c', 'd')); + expect(s.precision).toBe(0); + expect(s.recall).toBe(0); + expect(s.f1).toBe(null); // p+r==0 ⇒ harmonic mean undefined, reported n/a + expect(s.fnis).toEqual([k('c'), k('d')]); + }); + + it('empty CIS ⇒ precision n/a (null), recall 0, F1 n/a (the PDG-intra case)', () => { + // This is the SHAPE the real harness measures for PDG on a self-contained + // function: the mode reports nothing, AIS = {criterion}. precision is + // genuinely undefined (no predictions), recall is 0 (missed everything). + const s = M.score(new Set(), setOf('criterion')); + expect(s.precision).toBe(null); // |CIS|=0 ⇒ undefined, NOT 0 + expect(s.recall).toBe(0); + expect(s.f1).toBe(null); + expect(s.fnis).toEqual([k('criterion')]); // the dangerous miss + expect(s.cisAisRatio).toBe(0); + }); + + it('empty AIS ⇒ recall n/a (null) — a scope with no ground truth', () => { + const s = M.score(setOf('a'), new Set()); + expect(s.recall).toBe(null); // |AIS|=0 ⇒ undefined, NOT 0 + expect(s.precision).toBe(0); // predicted a, none real + expect(s.f1).toBe(null); + expect(s.cisAisRatio).toBe(null); + }); +}); + +describe('impact-pdg metric math — PDG line granularity (U7 rework)', () => { + it('pdgLineCis builds : keys from affectedStatements', () => { + const cis = M.pdgLineCis([ + { line: 10, filePath: 'src/a.ts', text: 'sum = sum + x;' }, + { line: 12, filePath: 'src/a.ts', text: 'return sum;' }, + // a malformed entry (no numeric line) is dropped, never keyed. + { filePath: 'src/a.ts', text: 'noise' }, + ]); + expect([...cis].sort()).toEqual(['src/a.ts:10', 'src/a.ts:12']); + }); + + it('intraLineAis builds line keys from intra_AIS; empty intra_AIS ⇒ empty set', () => { + const withLines = M.intraLineAis({ + intra_AIS: [ + { symbol: 'total', filePath: 'src/a.ts', line: 10 }, + { symbol: 'total', filePath: 'src/a.ts', line: 12 }, + ], + }); + expect([...withLines].sort()).toEqual(['src/a.ts:10', 'src/a.ts:12']); + // an inter fixture has empty intra_AIS ⇒ empty line AIS (recall n/a, not 0). + expect(M.intraLineAis({ intra_AIS: [] }).size).toBe(0); + }); + + it('scores a line slice that exactly matches intra_AIS ⇒ P=R=F1=1 (the accumulator)', () => { + // The verified accumulator case: line-8 slice returns {10,12} = intra_AIS. + const cis = M.pdgLineCis([ + { line: 10, filePath: 'src/accumulator.ts' }, + { line: 12, filePath: 'src/accumulator.ts' }, + ]); + const ais = M.intraLineAis({ + intra_AIS: [ + { filePath: 'src/accumulator.ts', line: 10 }, + { filePath: 'src/accumulator.ts', line: 12 }, + ], + }); + const s = M.score(cis, ais); + expect(s.precision).toBe(1); + expect(s.recall).toBe(1); + expect(s.f1).toBe(1); + }); + + it('an inter fixture (empty intra_AIS) ⇒ precision 0 on the router noise, recall n/a', () => { + // A pure-inter router's line slice returns its own routing returns as FPIS + // against the empty intra_AIS — the by-design "PDG is intra-procedural" case. + const cis = M.pdgLineCis([ + { line: 24, filePath: 'src/d.ts' }, + { line: 26, filePath: 'src/d.ts' }, + ]); + const s = M.score(cis, M.intraLineAis({ intra_AIS: [] })); + expect(s.precision).toBe(0); // 2 predicted, none in (empty) intra truth + expect(s.recall).toBe(null); // |AIS|=0 ⇒ recall n/a + expect(s.f1).toBe(null); + }); +}); + +describe('impact-pdg metric math — compareModes()', () => { + it('Jaccard + directional set-diffs split true/noise', () => { + // callgraph finds {a,b,c} (a,b real, c noise); pdg finds {b,d} (b real, d noise). + // AIS = {a,b,e}. + const cg = setOf('a', 'b', 'c'); + const pdg = setOf('b', 'd'); + const ais = setOf('a', 'b', 'e'); + const cmp = M.compareModes(cg, pdg, ais); + // union {a,b,c,d}=4, inter {b}=1 ⇒ Jaccard 1/4. + expect(cmp.jaccard).toBeCloseTo(0.25, 12); + expect(cmp.intersectionSize).toBe(1); + expect(cmp.unionSize).toBe(4); + // pdg-only = {d}; d ∉ AIS ⇒ noise. + expect(cmp.pdgOnly.all).toEqual([k('d')]); + expect(cmp.pdgOnly.true).toEqual([]); + expect(cmp.pdgOnly.noise).toEqual([k('d')]); + // callgraph-only = {a,c}; a ∈ AIS (true find pdg missed), c ∉ AIS (noise). + expect(cmp.callgraphOnly.all).toEqual([k('a'), k('c')]); + expect(cmp.callgraphOnly.true).toEqual([k('a')]); + expect(cmp.callgraphOnly.noise).toEqual([k('c')]); + }); + + it('two empty CIS ⇒ Jaccard n/a (null), no diffs', () => { + const cmp = M.compareModes(new Set(), new Set(), setOf('a')); + expect(cmp.jaccard).toBe(null); + expect(cmp.pdgOnly.all).toEqual([]); + expect(cmp.callgraphOnly.all).toEqual([]); + }); +}); + +describe('impact-pdg metric math — partitionCisByScope() / aisByScope()', () => { + it('partitions a CIS into intra (=criterion) vs inter (others)', () => { + const critKey = k('route', 'src/mixed.ts'); + const cis = M.toKeySet([ + k('route', 'src/mixed.ts'), // the criterion itself ⇒ intra + k('fast', 'src/mixed.ts'), // a callee ⇒ inter + k('slow', 'src/mixed.ts'), // a callee ⇒ inter + ]); + const part = M.partitionCisByScope(cis, critKey); + expect([...part.intra]).toEqual([critKey]); + expect([...part.inter].sort()).toEqual([k('fast', 'src/mixed.ts'), k('slow', 'src/mixed.ts')]); + expect(part.mixed.size).toBe(3); + }); + + it('aisByScope collapses intra_AIS lines onto the criterion symbol', () => { + const gt = { + criterion: { name: 'route', filePath: 'src/mixed.ts', direction: 'downstream' }, + intra_AIS: [ + { symbol: 'route', filePath: 'src/mixed.ts', line: 16 }, + { symbol: 'route', filePath: 'src/mixed.ts', line: 18 }, + { symbol: 'route', filePath: 'src/mixed.ts', line: 20 }, + ], + inter_AIS: [ + { symbol: 'fast', filePath: 'src/mixed.ts' }, + { symbol: 'slow', filePath: 'src/mixed.ts' }, + ], + }; + const a = M.aisByScope(gt); + // three intra lines collapse to the singleton {criterion}. + expect([...a.intra]).toEqual([k('route', 'src/mixed.ts')]); + expect([...a.inter].sort()).toEqual([k('fast', 'src/mixed.ts'), k('slow', 'src/mixed.ts')]); + expect(a.mixed.size).toBe(3); + }); + + it('aisByScope: empty intra_AIS ⇒ empty intra scope (no false {criterion})', () => { + const gt = { + criterion: { name: 'dispatch', filePath: 'src/d.ts', direction: 'downstream' }, + intra_AIS: [], + inter_AIS: [{ symbol: 'handleA', filePath: 'src/d.ts' }], + }; + const a = M.aisByScope(gt); + expect(a.intra.size).toBe(0); // no intra truth ⇒ recall will be n/a, not 0 + expect([...a.inter]).toEqual([k('handleA', 'src/d.ts')]); + }); +}); + +describe('impact-pdg metric math — unified axes', () => { + const gt = { + criterion: { name: 'route', filePath: 'src/mixed.ts', direction: 'downstream' }, + intra_AIS: [ + { symbol: 'route', filePath: 'src/mixed.ts', line: 16 }, + { symbol: 'route', filePath: 'src/mixed.ts', line: 18 }, + ], + inter_AIS: [ + { symbol: 'fast', filePath: 'src/mixed.ts' }, + { symbol: 'slow', filePath: 'src/mixed.ts' }, + ], + }; + + it('builds tagged unified AIS without mixing line and symbol keys', () => { + const ais = M.unifiedAis(gt); + expect([...ais.intraLine].sort()).toEqual([ + 'statement:src/mixed.ts:16', + 'statement:src/mixed.ts:18', + ]); + expect([...ais.interSymbol].sort()).toEqual([ + 'symbol:fast@src/mixed.ts', + 'symbol:slow@src/mixed.ts', + ]); + }); + + it('adapts current engines onto separate unified axes', () => { + const cg = M.callgraphUnifiedCis( + gt, + M.toKeySet([ + M.symbolKey('route', 'src/mixed.ts'), + M.symbolKey('fast', 'src/mixed.ts'), + M.symbolKey('slow', 'src/mixed.ts'), + ]), + ); + const pdg = M.pdgUnifiedCis( + M.pdgLineCis([ + { line: 16, filePath: 'src/mixed.ts' }, + { line: 18, filePath: 'src/mixed.ts' }, + ]), + ); + + expect([...cg.intraLine]).toEqual([]); + expect([...cg.interSymbol].sort()).toEqual([ + 'symbol:fast@src/mixed.ts', + 'symbol:slow@src/mixed.ts', + ]); + expect([...pdg.intraLine].sort()).toEqual([ + 'statement:src/mixed.ts:16', + 'statement:src/mixed.ts:18', + ]); + expect([...pdg.interSymbol]).toEqual([]); + }); + + it('adapts unified PDG onto both statement and inter-symbol axes', () => { + const pdg = M.pdgUnifiedCis( + M.pdgLineCis([ + { line: 16, filePath: 'src/mixed.ts' }, + { line: 18, filePath: 'src/mixed.ts' }, + ]), + M.toKeySet([ + M.symbolKey('route', 'src/mixed.ts'), + M.symbolKey('fast', 'src/mixed.ts'), + M.symbolKey('slow', 'src/mixed.ts'), + ]), + gt, + ); + + expect([...pdg.intraLine].sort()).toEqual([ + 'statement:src/mixed.ts:16', + 'statement:src/mixed.ts:18', + ]); + expect([...pdg.interSymbol].sort()).toEqual([ + 'symbol:fast@src/mixed.ts', + 'symbol:slow@src/mixed.ts', + ]); + + const scored = M.scoreUnifiedAxes(pdg, M.unifiedAis(gt)); + expect(scored.intraLine.f1).toBe(1); + expect(scored.interSymbol.f1).toBe(1); + }); + + it('scores composed-current as exact on both axes without a blended F1', () => { + const ais = M.unifiedAis(gt); + const cg = M.callgraphUnifiedCis( + gt, + M.toKeySet([M.symbolKey('fast', 'src/mixed.ts'), M.symbolKey('slow', 'src/mixed.ts')]), + ); + const pdg = M.pdgUnifiedCis( + M.pdgLineCis([ + { line: 16, filePath: 'src/mixed.ts' }, + { line: 18, filePath: 'src/mixed.ts' }, + ]), + ); + + const composed = M.composeUnifiedCis(cg, pdg); + const scored = M.scoreUnifiedAxes(composed, ais); + + expect(scored.intraLine.f1).toBe(1); + expect(scored.interSymbol.f1).toBe(1); + + const agg = M.aggregateUnifiedScores([scored]); + expect(agg.intraLine.f1).toBe(1); + expect(agg.interSymbol.f1).toBe(1); + expect(agg.minRecall).toBe(1); + expect(agg.fpis).toBe(0); + expect(agg.fnis).toBe(0); + expect(agg).not.toHaveProperty('f1'); + }); + + it('makes current standalone engines visibly incomplete on one unified axis', () => { + const ais = M.unifiedAis(gt); + const cg = M.callgraphUnifiedCis(gt, M.toKeySet([M.symbolKey('fast', 'src/mixed.ts')])); + const pdg = M.pdgUnifiedCis(M.pdgLineCis([{ line: 16, filePath: 'src/mixed.ts' }])); + + const cgScore = M.scoreUnifiedAxes(cg, ais); + const pdgScore = M.scoreUnifiedAxes(pdg, ais); + + expect(cgScore.intraLine.recall).toBe(0); + expect(cgScore.interSymbol.recall).toBe(0.5); + expect(pdgScore.intraLine.recall).toBe(0.5); + expect(pdgScore.interSymbol.recall).toBe(0); + }); +}); + +describe('impact-pdg metric math — aggregate()', () => { + it('macro-averages defined metrics, EXCLUDING nulls (not folding as 0)', () => { + const per = [ + { precision: 1, recall: 1, f1: 1, cisAisRatio: 1, fpisCount: 0, fnisCount: 0 }, + { precision: 0.5, recall: 1, f1: 2 / 3, cisAisRatio: 2, fpisCount: 1, fnisCount: 0 }, + // a null-precision case (|CIS|=0): excluded from the precision mean. + { precision: null, recall: 0, f1: null, cisAisRatio: 0, fpisCount: 0, fnisCount: 2 }, + ]; + const agg = M.aggregate(per); + expect(agg.nCases).toBe(3); + // precision mean over the 2 defined cases = (1+0.5)/2 = 0.75 + expect(agg.precision).toBeCloseTo(0.75, 12); + expect(agg.nPrecision).toBe(2); + // recall mean over all 3 (none null) = (1+1+0)/3 + expect(agg.recall).toBeCloseTo(2 / 3, 12); + expect(agg.nRecall).toBe(3); + // F1 mean over the 2 defined = (1 + 2/3)/2 + expect(agg.f1).toBeCloseTo((1 + 2 / 3) / 2, 12); + expect(agg.nF1).toBe(2); + expect(agg.fpis).toBe(1); // summed totals + expect(agg.fnis).toBe(2); + }); + + it('all-null stratum ⇒ null means, n=0 (reported n/a)', () => { + const agg = M.aggregate([{ precision: null, recall: null, f1: null, cisAisRatio: null }]); + expect(agg.precision).toBe(null); + expect(agg.recall).toBe(null); + expect(agg.f1).toBe(null); + expect(agg.nF1).toBe(0); + }); +}); + +describe('impact-pdg metric math — annotation fingerprint (KTD10)', () => { + const fakeHash = (s: string): string => { + // tiny deterministic non-crypto digest — enough to assert drift sensitivity + // without pulling node:crypto into the unit (the real harness injects sha256). + let h = 5381; + for (let i = 0; i < s.length; i++) h = ((h << 5) + h + s.charCodeAt(i)) >>> 0; + return h.toString(16); + }; + const fx = (over: Record = {}) => ({ + name: 'c1', + gt: { + schemaVersion: 1, + criterion: { + name: 'f', + filePath: 'src/f.ts', + direction: 'downstream', + marker: 'x', + pdgEdgeKinds: ['REACHING_DEF'], + }, + locus: 'intra', + provenance: 'manual', + intra_AIS: [{ symbol: 'f', filePath: 'src/f.ts', line: 3 }], + inter_AIS: [], + ...over, + }, + }); + + it('is order-independent over the fixture list', () => { + const a = M.fingerprintAnnotationSet([fx({}), { ...fx({}), name: 'c2' }], fakeHash); + const b = M.fingerprintAnnotationSet([{ ...fx({}), name: 'c2' }, fx({})], fakeHash); + expect(a).toBe(b); + }); + + it('trips when an AIS membership changes (catches unreviewed ground-truth edits)', () => { + const base = M.fingerprintAnnotationSet([fx({})], fakeHash); + const edited = M.fingerprintAnnotationSet( + [fx({ intra_AIS: [{ symbol: 'f', filePath: 'src/f.ts', line: 99 }] })], + fakeHash, + ); + expect(edited).not.toBe(base); + }); + + it('trips when criterion.line changes (the PDG slice seed — U7 rework)', () => { + // criterion.line is part of the ground truth: it seeds the statement-anchored + // PDG slice, so changing it changes the measured impact set and MUST trip. + const base = M.fingerprintAnnotationSet( + [ + fx({ + criterion: { + name: 'f', + filePath: 'src/f.ts', + direction: 'downstream', + line: 8, + marker: 'x', + pdgEdgeKinds: ['REACHING_DEF'], + }, + }), + ], + fakeHash, + ); + const moved = M.fingerprintAnnotationSet( + [ + fx({ + criterion: { + name: 'f', + filePath: 'src/f.ts', + direction: 'downstream', + line: 9, + marker: 'x', + pdgEdgeKinds: ['REACHING_DEF'], + }, + }), + ], + fakeHash, + ); + expect(moved).not.toBe(base); + }); + + it('trips when the criterion direction flips', () => { + const base = M.fingerprintAnnotationSet([fx({})], fakeHash); + const flipped = M.fingerprintAnnotationSet( + [ + fx({ + criterion: { + name: 'f', + filePath: 'src/f.ts', + direction: 'upstream', + marker: 'x', + pdgEdgeKinds: ['REACHING_DEF'], + }, + }), + ], + fakeHash, + ); + expect(flipped).not.toBe(base); + }); + + it('is STABLE under a pure reordering of AIS entries within a case', () => { + const a = M.fingerprintAnnotationSet( + [ + fx({ + intra_AIS: [ + { symbol: 'f', filePath: 'src/f.ts', line: 3 }, + { symbol: 'f', filePath: 'src/f.ts', line: 5 }, + ], + }), + ], + fakeHash, + ); + const b = M.fingerprintAnnotationSet( + [ + fx({ + intra_AIS: [ + { symbol: 'f', filePath: 'src/f.ts', line: 5 }, + { symbol: 'f', filePath: 'src/f.ts', line: 3 }, + ], + }), + ], + fakeHash, + ); + expect(a).toBe(b); + }); +}); + +describe('impact-pdg metric math — median (substrate-stability gate F5)', () => { + it('odd/even/empty', () => { + expect(M.median([3, 1, 2])).toBe(2); + expect(M.median([4, 1, 3, 2])).toBe(2.5); + expect(M.median([])).toBe(null); + }); +}); + +describe('impact-pdg metric math — U2 mutation/dynamic-oracle scorers', () => { + const lk = (line: number, file = 'src/a.ts'): string => `${file}:${line}`; + + it('mutationRecall = |B ∩ slice| / |B|, with missing (B∖slice) and extra (slice∖B)', () => { + // B (dynamic AIS the oracle proved) = {8,9,10}; slice (static PDG) = {9,10,11}. + // ∩ = {9,10} = 2. recall = 2/3. missing = {8} (a static recall hole). extra = {11}. + const B = new Set([lk(8), lk(9), lk(10)]); + const slice = new Set([lk(9), lk(10), lk(11)]); + const r = M.mutationRecall(B, slice); + expect(r.recall).toBeCloseTo(2 / 3, 12); + expect(r.bSize).toBe(3); + expect(r.sliceSize).toBe(3); + expect(r.intersection).toBe(2); + expect(r.missing).toEqual([lk(8)]); // B ∖ slice — the dangerous miss + expect(r.extra).toEqual([lk(11)]); // slice ∖ B — sound over-approximation + }); + + it('mutationRecall = 1.0 when the static slice covers every proven line', () => { + const r = M.mutationRecall(new Set([lk(10), lk(12)]), new Set([lk(10), lk(11), lk(12)])); + expect(r.recall).toBe(1); + expect(r.missing).toEqual([]); + expect(r.extra).toEqual([lk(11)]); // extra is informational, not gated + }); + + it('mutationRecall on empty B ⇒ recall null (nothing proven to find), never 0', () => { + const r = M.mutationRecall(new Set(), new Set([lk(10)])); + expect(r.recall).toBe(null); + expect(r.bSize).toBe(0); + expect(r.missing).toEqual([]); + }); + + it('mutationRecall accepts plain arrays as well as Sets', () => { + const r = M.mutationRecall([lk(1), lk(2)], [lk(2)]); + expect(r.recall).toBe(0.5); + expect(r.missing).toEqual([lk(1)]); + }); + + it('circularityDiff: beyondManual (B∖M) is the independent annotation-gap evidence', () => { + // Oracle proved {8,9,10}; manual intra_AIS = {9,10}. B∖M = {8} ⇒ the manual + // annotation missed line 8 (the headline circularity signal). confirmed = {9,10}. + const c = M.circularityDiff(new Set([lk(8), lk(9), lk(10)]), new Set([lk(9), lk(10)])); + expect(c.beyondManual).toEqual([lk(8)]); + // confirmed (B ∩ M) is lexically sorted: "src/a.ts:10" sorts before "src/a.ts:9". + expect(c.confirmed).toEqual([lk(10), lk(9)]); + expect(c.manualOnly).toEqual([]); + }); + + it('circularityDiff: empty beyondManual ⇒ the annotation independently confirmed', () => { + const c = M.circularityDiff(new Set([lk(9), lk(10)]), new Set([lk(9), lk(10), lk(11)])); + expect(c.beyondManual).toEqual([]); + expect(c.confirmed).toEqual([lk(10), lk(9)]); // lexically sorted + expect(c.manualOnly).toEqual([lk(11)]); // manual claimed 11; oracle did not prove it + }); + + it('isEquivalentMutant: empty behavioral set ⇒ equivalent (discarded from the union)', () => { + expect(M.isEquivalentMutant({ diffLines: [] })).toBe(true); + expect(M.isEquivalentMutant({ diffLines: [lk(9)] })).toBe(false); + expect(M.isEquivalentMutant([])).toBe(true); + expect(M.isEquivalentMutant([lk(9)])).toBe(false); + expect(M.isEquivalentMutant(new Set())).toBe(true); + }); + + it('mutation fingerprint is order-independent over fixtures and trips on a proven-set change', () => { + const fakeHash = (s: string): string => { + let h = 5381; + for (let i = 0; i < s.length; i++) h = ((h << 5) + h + s.charCodeAt(i)) >>> 0; + return h.toString(16); + }; + const fA = { + name: 'a', + criterionKey: 'src/a.ts:7', + behavioralAis: [lk(8), lk(9)], + mutants: [{ op: 'AOR', diffLines: [lk(8)] }], + }; + const fB = { + name: 'b', + criterionKey: 'src/b.ts:5', + behavioralAis: [lk(6, 'src/b.ts')], + mutants: [{ op: 'ROR', diffLines: [lk(6, 'src/b.ts')] }], + }; + const fwd = M.fingerprintMutationSet([fA, fB], fakeHash); + const rev = M.fingerprintMutationSet([fB, fA], fakeHash); + expect(fwd).toBe(rev); // order-independent over the fixture list + const changed = M.fingerprintMutationSet( + [{ ...fA, behavioralAis: [lk(8), lk(9), lk(99)] }, fB], + fakeHash, + ); + expect(changed).not.toBe(fwd); // a change in what the oracle proves trips it + }); + + it('mutation fingerprint ignores EQUIVALENT mutants (empty diffLines carry no signal)', () => { + const fakeHash = (s: string): string => String(s.length); + const base = M.canonicalizeMutationSet([ + { + name: 'a', + criterionKey: 'src/a.ts:7', + behavioralAis: [lk(8)], + mutants: [{ op: 'AOR', diffLines: [lk(8)] }], + }, + ]); + const withEquiv = M.canonicalizeMutationSet([ + { + name: 'a', + criterionKey: 'src/a.ts:7', + behavioralAis: [lk(8)], + mutants: [ + { op: 'AOR', diffLines: [lk(8)] }, + { op: 'CRP', diffLines: [] }, // equivalent — must not change the canonical op set + ], + }, + ]); + expect(withEquiv).toBe(base); + }); +}); diff --git a/gitnexus/test/unit/impact-pdg-real-code-metrics.test.ts b/gitnexus/test/unit/impact-pdg-real-code-metrics.test.ts new file mode 100644 index 000000000..cbf3a93db --- /dev/null +++ b/gitnexus/test/unit/impact-pdg-real-code-metrics.test.ts @@ -0,0 +1,132 @@ +import { describe, expect, it } from 'vitest'; + +import { + compareSymbolSets, + evaluateCheckGates, + median, + percentile, + summarizeCases, + symbolKeysFromByDepth, +} from '../../bench/impact-pdg/real-code.mjs'; + +describe('impact-pdg real-code metric helpers', () => { + it('extracts stable symbol keys from byDepth records', () => { + const keys = symbolKeysFromByDepth({ + 1: [ + { id: 'Function:src/a.ts:a', name: 'a', filePath: 'src/a.ts' }, + { id: null, name: 'dynamicTarget', filePath: 'src/b.ts' }, + ], + }); + + expect([...keys].sort()).toEqual(['Function:src/a.ts:a', 'dynamicTarget@src/b.ts']); + }); + + it('compares candidate symbol reach against a reference set', () => { + const comparison = compareSymbolSets(new Set(['A', 'B']), new Set(['A', 'C'])); + + expect(comparison).toMatchObject({ + referenceSize: 2, + candidateSize: 2, + overlapSize: 1, + recallVsReference: 0.5, + precisionVsReference: 0.5, + jaccard: 1 / 3, + referenceOnly: ['B'], + candidateOnly: ['C'], + }); + }); + + it('summarizes latency and quality proxy metrics across cases', () => { + const summary = summarizeCases([ + { + latencyMs: { + callgraph: { median: 10 }, + pdg: { median: 25, p95: 30 }, + pdgOverCallgraphMedian: 2.5, + }, + callgraph: { error: null, partial: false }, + pdg: { + error: null, + pdgLayer: 'ready', + partial: false, + epistemic: 'pdg-intra-procedural', + evidenceCounts: { 'callgraph-bridge': 2 }, + }, + symbolAgreement: { + recallVsReference: 1, + precisionVsReference: 1, + }, + }, + { + latencyMs: { + callgraph: { median: 20 }, + pdg: { median: 40, p95: 45 }, + pdgOverCallgraphMedian: 2, + }, + callgraph: { error: null, partial: false }, + pdg: { + error: null, + pdgLayer: 'ready', + partial: true, + epistemic: 'pdg-intra-procedural', + evidenceCounts: { 'unproven-bridge': 1 }, + }, + symbolAgreement: { + recallVsReference: 0.5, + precisionVsReference: 1, + }, + }, + ] as any); + + expect(summary.performance).toMatchObject({ + callgraphMedianMs: 15, + pdgMedianMs: 32.5, + pdgP95Ms: 45, + pdgOverCallgraphMedian: 2.25, + }); + expect(summary.qualityProxy).toMatchObject({ + comparableCases: 2, + meanSymbolRecallVsCallgraph: 0.75, + minSymbolRecallVsCallgraph: 0.5, + meanSymbolPrecisionVsCallgraph: 1, + degradedCaseCount: 0, + errorCaseCount: 0, + partialCaseCount: 1, + evidenceCounts: { 'callgraph-bridge': 2, 'unproven-bridge': 1 }, + unprovenBridgeRatio: 0.333, + }); + }); + + it('reports explicit check failures for degraded quality and slow PDG medians', () => { + const failures = evaluateCheckGates( + { + cases: [{}], + summary: { + performance: { pdgMedianMs: 600 }, + qualityProxy: { + errorCaseCount: 1, + degradedCaseCount: 1, + minSymbolRecallVsCallgraph: 0.5, + }, + }, + } as any, + { + GN_REAL_CODE_PDG_MIN_SYMBOL_RECALL: '0.9', + GN_REAL_CODE_PDG_MAX_MEDIAN_MS: '500', + } as any, + ); + + expect(failures).toEqual([ + '1 case(s) returned errors', + '1 case(s) reported a degraded PDG layer', + 'min PDG symbol recall vs callgraph 0.5 < 0.9', + 'PDG median latency 600ms > 500ms', + ]); + }); + + it('computes median and percentile deterministically', () => { + expect(median([9, 1, 3])).toBe(3); + expect(median([9, 1, 3, 5])).toBe(4); + expect(percentile([10, 30, 20], 95)).toBe(30); + }); +}); diff --git a/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts index a36b6b046..679dab52c 100644 --- a/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts +++ b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts @@ -107,9 +107,12 @@ describe('buildPhaseList parity (registry refactor, #2080)', () => { // inserted right after pruneLocalSymbols, before mro. // --------------------------------------------------------------------------- +// pdg-gated phases are registered consecutively after pruneLocalSymbols, in +// registration order: taintSummaries (#2084) then callSummaries (PDG FU-C). const WITH_TAINT_SUMMARIES = [ ...FULL_ORDER.slice(0, FULL_ORDER.indexOf('pruneLocalSymbols') + 1), 'taintSummaries', + 'callSummaries', ...FULL_ORDER.slice(FULL_ORDER.indexOf('pruneLocalSymbols') + 1), ]; @@ -130,12 +133,19 @@ describe('buildPhaseList — taintSummaries opt-in (#2084)', () => { expect(names).not.toContain('mro'); }); - it('no always-on phase depends on the pdg-gated taintSummaries phase', () => { + it('no always-on phase depends on the pdg-gated taintSummaries/callSummaries phases', () => { // A filtered-out dep would throw in getPhaseOutput at runtime, so no - // always-included phase may list taintSummaries in its deps. + // always-included phase may list a pdg-gated phase in its deps. const offList = buildPhaseList({}); for (const p of offList) { expect(p.deps).not.toContain('taintSummaries'); + expect(p.deps).not.toContain('callSummaries'); } }); + + it('pdg:true → callSummaries inserted alongside taintSummaries, absent when pdg off', () => { + expect(buildPhaseList({}).map((p) => p.name)).not.toContain('callSummaries'); + expect(buildPhaseList({ pdg: false }).map((p) => p.name)).not.toContain('callSummaries'); + expect(buildPhaseList({ pdg: true }).map((p) => p.name)).toContain('callSummaries'); + }); }); diff --git a/gitnexus/test/unit/pdg-bridge-id-match.test.ts b/gitnexus/test/unit/pdg-bridge-id-match.test.ts new file mode 100644 index 000000000..3f3374f44 --- /dev/null +++ b/gitnexus/test/unit/pdg-bridge-id-match.test.ts @@ -0,0 +1,177 @@ +/** + * U5 — Bridge: id-match primary with leaf-name fallback. + * + * `pdgBridgeEvidenceForImpact` proves a callgraph-reached first-hop callee by its + * RESOLVED SYMBOL ID against the slice's `calleeIds` (sound), falling back to the + * leaf-name predicate ONLY when ids are absent/empty or the block is capped + * (sentinel). These cases assert the id path eliminates the same-leaf-name + * collision FP (R1) and the import-alias FN (R1), that an id-miss is a real proof + * failure rather than a name fall-through (KTD3), and that the R3/R7 fallbacks are + * preserved verbatim. No `if`-branching; unconditional `toMatchObject`/`toBe`. + */ +import { describe, it, expect } from 'vitest'; +import { pdgBridgeEvidenceForImpact } from '../../src/mcp/local/pdg-impact.js'; +import { CALLEES_TRUNCATED_SENTINEL } from '../../src/core/ingestion/cfg/emit.js'; + +describe('pdgBridgeEvidenceForImpact — U5 resolved-id match (KTD3)', () => { + it('eliminates the same-leaf-name collision FP: only the on-slice id is proven (R1)', () => { + // Two reached callees share the leaf name "get" but have distinct resolved ids + // (idA on the slice, idB not). The NAME path with sliceCalleeNames={"get"} + // would prove BOTH; the id path proves only idA. + const sliceCalleeIds = new Set(['idA']); + const onSlice = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds, sliceCalleeNames: new Set(['get']) }, + depth: 1, + calleeName: 'get', + calleeId: 'idA', + }); + const offSlice = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds, sliceCalleeNames: new Set(['get']) }, + depth: 1, + calleeName: 'get', + calleeId: 'idB', + }); + expect(onSlice).toMatchObject({ evidence: 'callgraph-bridge' }); + expect(offSlice).toMatchObject({ evidence: 'unproven-bridge' }); + }); + + it('eliminates the import-alias FN: proven by id though the name would miss (R1)', () => { + // The reached callee's leaf name "bar" (an alias/rename) is NOT in + // sliceCalleeNames, but its resolved id idFoo IS in sliceCalleeIds — the id + // path proves it where the name path would drop it. + const result = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds: new Set(['idFoo']), sliceCalleeNames: new Set(['foo']) }, + depth: 1, + calleeName: 'bar', + calleeId: 'idFoo', + }); + expect(result).toMatchObject({ evidence: 'callgraph-bridge' }); + }); + + it('id-miss is unproven, NOT a name fall-through (KTD3 proof failure)', () => { + // ids present + non-empty, sentinel absent: a reached id not in the set is a + // proof failure even when its leaf name DOES match — the matching name must + // not rescue it (that would re-leak the collision the id key eliminates). + const result = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds: new Set(['idA']), sliceCalleeNames: new Set(['get']) }, + depth: 1, + calleeName: 'get', + calleeId: 'idZ', + }); + expect(result).toMatchObject({ evidence: 'unproven-bridge' }); + expect(result.evidence).not.toBe('callgraph-bridge'); + }); + + it('falls back to the name path when ids are absent (R3) — current behavior preserved', () => { + const result = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeNames: new Set(['get']), sliceCalleeIds: new Set() }, + depth: 1, + calleeName: 'get', + }); + expect(result).toMatchObject({ evidence: 'callgraph-bridge' }); + }); + + it('falls back to the name path when the id set is empty (not all-unproven)', () => { + // An empty sliceCalleeIds means "no captured ids", which must route to the + // name fallback — not collapse every reach to unproven. + const result = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds: new Set(), sliceCalleeNames: new Set(['get']) }, + depth: 1, + calleeName: 'get', + calleeId: 'idZ', + }); + expect(result).toMatchObject({ evidence: 'callgraph-bridge' }); + }); + + it('a capped block (sentinel) stays callgraph-equal regardless of ids (R7)', () => { + // The truncation sentinel marks the callee set incomplete; the id path is + // skipped (callee-unknown) so even an id-miss stays callgraph-bridge. + const result = pdgBridgeEvidenceForImpact({ + bridge: { + sliceCalleeIds: new Set(['idA']), + sliceCalleeNames: new Set([CALLEES_TRUNCATED_SENTINEL, 'get']), + }, + depth: 1, + calleeName: 'get', + calleeId: 'idZ', + }); + expect(result).toMatchObject({ evidence: 'callgraph-bridge' }); + }); + + it('multi-target dispatch: a reached id in a multi-id set is proven (R2)', () => { + const result = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds: new Set(['idA', 'idB']), sliceCalleeNames: new Set() }, + depth: 1, + calleeName: 'dispatch', + calleeId: 'idB', + }); + expect(result).toMatchObject({ evidence: 'callgraph-bridge' }); + }); + + // PR #2227 tri-review-2 headline: an id-only slice (sliceCalleeNames empty/absent, + // sliceCalleeIds present) must id-DISCRIMINATE, not short-circuit to "prove + // everything" via the whole-symbol guard. The `basis` distinguishes the id-match + // path from the whole-symbol short-circuit. + it('id-only slice, id IN set → proven via the id path (not whole-symbol)', () => { + const result = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds: new Set(['idA']), sliceCalleeNames: new Set() }, // names empty (id-only index) + depth: 1, + calleeName: 'get', + calleeId: 'idA', + }); + expect(result).toMatchObject({ + evidence: 'callgraph-bridge', + basis: + 'callee id is invoked in a block of the local PDG dependence slice (resolved-symbol match)', + }); + }); + + it('id-only slice, id NOT in set → unproven-bridge (the over-prove bug) and does not throw', () => { + // Before the fix this returned callgraph-bridge (over-prove): the empty-names + // guard short-circuited before the id branch. It must now id-discriminate, and + // the sentinel/name reads operate on an empty names set without throwing. + const result = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds: new Set(['idA']), sliceCalleeNames: new Set() }, // names empty (id-only index) + depth: 1, + calleeName: 'get', + calleeId: 'idZ', + }); + expect(result).toMatchObject({ + evidence: 'unproven-bridge', + basis: 'callee id is not invoked in any block of the local PDG dependence slice', + }); + }); + + it('both keys empty → whole-symbol compatibility bridge (unchanged)', () => { + const result = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeNames: new Set(), sliceCalleeIds: new Set() }, + depth: 1, + calleeName: 'get', + calleeId: 'idZ', + }); + expect(result).toMatchObject({ + evidence: 'callgraph-bridge', + basis: 'whole-symbol PDG result uses symbol graph as compatibility bridge', + }); + }); + + it('depth>1 inherited evidence is unchanged by the id path', () => { + const inherited = { evidence: 'callgraph-bridge' as const, basis: 'inherited proven' }; + const result = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds: new Set(['idA']), sliceCalleeNames: new Set(['get']) }, + depth: 2, + calleeName: 'get', + calleeId: 'idZ', + inherited, + }); + expect(result).toMatchObject(inherited); + // No inherited supplied → the documented depth>1 default, not an id verdict. + const fallback = pdgBridgeEvidenceForImpact({ + bridge: { sliceCalleeIds: new Set(['idA']), sliceCalleeNames: new Set() }, + depth: 2, + calleeName: 'get', + calleeId: 'idZ', + }); + expect(fallback).toMatchObject({ evidence: 'unproven-bridge' }); + }); +}); diff --git a/gitnexus/test/unit/pdg-callee-id-capture.test.ts b/gitnexus/test/unit/pdg-callee-id-capture.test.ts new file mode 100644 index 000000000..6e769ecb0 --- /dev/null +++ b/gitnexus/test/unit/pdg-callee-id-capture.test.ts @@ -0,0 +1,624 @@ +/** + * Unit tests for the resolved-callee-id capture sink (#2227 follow-up plan U2). + * + * During Phase-4 scope-resolution CALLS-edge emission, each resolved call site's + * `(line, col) → calleeId` is accumulated across ALL THREE CALLS emit paths, + * each BEFORE its dedup (KTD6/R8), gated on `--pdg`. A later unit (U3) joins + * this to CFG BasicBlocks by exact call-site position. + * + * Coordinate base (KTD7 — load-bearing): the sink keys on + * `atRange.startLine` / `atRange.startCol`, which are 1-based line / 0-based col + * (`nodeToCapture` builds them as `row + 1` / `column`; the `Range` doc confirms + * "1-based startLine; 0-based startCol"). This is byte-equal to U1's + * `SiteRecord.at` (`[startPosition.row + 1, startPosition.column]`), so the U3 + * position join lands. + * + * Strategy: + * - `tryEmitEdge` / `tryEmitEdgeWithExplicitTargetId` and `emitReferencesViaLookup` + * are driven directly with a real `ScopeResolutionIndexes` + `GraphNodeLookup` + * (the emit-references.test.ts fixture pattern), so the capture runs on the + * real emit path. + * - `emitFreeCallFallback` is driven with a hand-built but fully-typed real + * `ParsedFile` whose Module-scope bindings resolve a free call — exercising + * the inline `addRelationship` capture line (the regression guard for the + * "only tryEmitEdge" bug). + */ + +import { describe, it, expect } from 'vitest'; +import { + buildDefIndex, + buildMethodDispatchIndex, + buildModuleScopeIndex, + buildQualifiedNameIndex, + buildScopeTree, + type BindingRef, + type NodeLabel, + type ParsedFile, + type Range, + type Reference, + type ReferenceSite, + type Scope, + type ScopeId, + type SymbolDefinition, +} from 'gitnexus-shared'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import type { KnowledgeGraph } from '../../src/core/graph/types.js'; +import type { ScopeResolutionIndexes } from '../../src/core/ingestion/model/scope-resolution-indexes.js'; +import { + buildGraphNodeLookup, + type GraphNodeLookup, +} from '../../src/core/ingestion/scope-resolution/graph-bridge/node-lookup.js'; +import { + tryEmitEdge, + tryEmitEdgeWithExplicitTargetId, +} from '../../src/core/ingestion/scope-resolution/graph-bridge/edges.js'; +import { emitReferencesViaLookup } from '../../src/core/ingestion/scope-resolution/graph-bridge/references-to-edges.js'; +import { emitFreeCallFallback } from '../../src/core/ingestion/scope-resolution/passes/free-call-fallback.js'; +import { buildWorkspaceResolutionIndex } from '../../src/core/ingestion/scope-resolution/workspace-index.js'; +import { createSemanticModel } from '../../src/core/ingestion/model/semantic-model.js'; +import { + createCalleeIdAccumulator, + calleeIdPosKey, + type CalleeIdAccumulator, +} from '../../src/core/ingestion/scope-resolution/graph-bridge/callee-id-sink.js'; + +// ─── Fixture builders ───────────────────────────────────────────────────── + +const FILE = 'x.ts'; + +const range = (sl = 1, sc = 0, el = 100, ec = 0): Range => ({ + startLine: sl, + startCol: sc, + endLine: el, + endCol: ec, +}); + +const def = ( + nodeId: string, + type: SymbolDefinition['type'] = 'Function', + qname?: string, + filePath = FILE, +): SymbolDefinition => ({ + nodeId, + filePath, + type, + ...(qname !== undefined ? { qualifiedName: qname } : {}), +}); + +const scope = ( + id: ScopeId, + parent: ScopeId | null, + kind: Scope['kind'], + ownedDefs: readonly SymbolDefinition[] = [], + r: Range = range(), + filePath = FILE, + bindings: Record = {}, +): Scope => ({ + id, + parent, + kind, + range: r, + filePath, + bindings: new Map(Object.entries(bindings)), + ownedDefs, + imports: [], + typeBindings: new Map(), +}); + +function makeIndexes( + scopes: readonly Scope[], + allDefs: readonly SymbolDefinition[], +): ScopeResolutionIndexes { + return { + scopeTree: buildScopeTree([...scopes]), + defs: buildDefIndex([...allDefs]), + qualifiedNames: buildQualifiedNameIndex([...allDefs]), + moduleScopes: buildModuleScopeIndex( + scopes + .filter((s) => s.kind === 'Module') + .map((s) => ({ filePath: s.filePath, moduleScopeId: s.id })), + ), + methodDispatch: buildMethodDispatchIndex({ + owners: [], + computeMro: () => [], + implementsOf: () => [], + }), + imports: new Map(), + bindings: new Map(), + bindingAugmentations: new Map(), + workspaceFqnBindings: new Map(), + workspaceTypeBindings: new Map(), + namespaceFqnBindings: new Map(), + namespaceTypeBindings: new Map(), + accessibleNamespacesByScope: new Map(), + referenceSites: [], + sccs: [], + stats: { + totalFiles: 0, + totalEdges: 0, + linkedEdges: 0, + unresolvedEdges: 0, + sccCount: 0, + largestSccSize: 0, + }, + }; +} + +/** A graph node for a Function so `buildGraphNodeLookup` registers it. */ +function fnNode(graph: KnowledgeGraph, id: string, name: string, filePath = FILE): void { + graph.addNode({ + id, + label: 'Function' as NodeLabel, + properties: { name, filePath, qualifiedName: name }, + }); +} + +/** A call-kind reference site at a given position, for the receiver-bound / + * direct `tryEmitEdge` driver. The bridge reads `inScope`, `atRange`, `kind`. */ +function callSite(inScope: ScopeId, line: number, col: number): ReferenceSite { + return { + name: 'callee', + atRange: range(line, col, line, col + 4), + inScope, + kind: 'call', + }; +} + +/** Collapse `accumulator.get(file)` into a plain `{ posKey: sortedIds[] }` + * object for unconditional `toEqual` / `toMatchObject` assertions. */ +function snapshot(acc: CalleeIdAccumulator, filePath: string): Record { + const byPos = acc.get(filePath); + const out: Record = {}; + for (const [key, ids] of byPos ?? new Map>()) { + out[key] = [...ids].sort(); + } + return out; +} + +// ─── Path 1: tryEmitEdge ────────────────────────────────────────────────── + +describe('callee-id capture — tryEmitEdge (receiver-bound path)', () => { + it('captures two receiver-bound CALLS at distinct positions (pos → {id})', () => { + const callerFn = def('def:caller', 'Function', 'caller'); + const targetA = def('def:targetA', 'Function', 'targetA'); + const targetB = def('def:targetB', 'Function', 'targetB'); + const mod = scope('scope:m', null, 'Module', [callerFn, targetA, targetB]); + const indexes = makeIndexes([mod], [callerFn, targetA, targetB]); + + const graph = createKnowledgeGraph(); + fnNode(graph, 'fn:caller', 'caller'); + fnNode(graph, 'fn:targetA', 'targetA'); + fnNode(graph, 'fn:targetB', 'targetB'); + const lookup: GraphNodeLookup = buildGraphNodeLookup(graph); + + const acc = createCalleeIdAccumulator(); + const seen = new Set(); + const okA = tryEmitEdge( + graph, + indexes, + lookup, + callSite('scope:m', 10, 4), + targetA, + 'call', + seen, + 0.85, + false, + { sink: acc, filePath: FILE }, + ); + const okB = tryEmitEdge( + graph, + indexes, + lookup, + callSite('scope:m', 20, 8), + targetB, + 'call', + seen, + 0.85, + false, + { sink: acc, filePath: FILE }, + ); + + expect(okA).toBe(true); + expect(okB).toBe(true); + expect(snapshot(acc, FILE)).toEqual({ + [calleeIdPosKey(10, 4)]: ['fn:targetA'], + [calleeIdPosKey(20, 8)]: ['fn:targetB'], + }); + }); + + it('R2 dispatch — one site, two resolved targets → pos → {idA, idB}', () => { + const callerFn = def('def:caller', 'Function', 'caller'); + const targetA = def('def:dispA', 'Function', 'dispA'); + const targetB = def('def:dispB', 'Function', 'dispB'); + const mod = scope('scope:m', null, 'Module', [callerFn, targetA, targetB]); + const indexes = makeIndexes([mod], [callerFn, targetA, targetB]); + + const graph = createKnowledgeGraph(); + fnNode(graph, 'fn:caller', 'caller'); + fnNode(graph, 'fn:dispA', 'dispA'); + fnNode(graph, 'fn:dispB', 'dispB'); + const lookup = buildGraphNodeLookup(graph); + + const acc = createCalleeIdAccumulator(); + const seen = new Set(); + // Same site (same position) dispatched to two distinct targets — mirrors + // interface-dispatch emitting a secondary CALLS edge for one call site. + const site = callSite('scope:m', 30, 2); + tryEmitEdge(graph, indexes, lookup, site, targetA, 'call', seen, 0.85, false, { + sink: acc, + filePath: FILE, + }); + tryEmitEdge(graph, indexes, lookup, site, targetB, 'interface-dispatch', seen, 0.85, false, { + sink: acc, + filePath: FILE, + }); + + expect(snapshot(acc, FILE)).toEqual({ + [calleeIdPosKey(30, 2)]: ['fn:dispA', 'fn:dispB'], + }); + }); + + it('dedup-independence — same target, two lines, collapse on → both positions captured', () => { + const callerFn = def('def:caller', 'Function', 'caller'); + const target = def('def:target', 'Function', 'target'); + const mod = scope('scope:m', null, 'Module', [callerFn, target]); + const indexes = makeIndexes([mod], [callerFn, target]); + + const graph = createKnowledgeGraph(); + fnNode(graph, 'fn:caller', 'caller'); + fnNode(graph, 'fn:target', 'target'); + const lookup = buildGraphNodeLookup(graph); + + const acc = createCalleeIdAccumulator(); + const seen = new Set(); + // collapse = true ⇒ the dedup key drops the line, so the SECOND edge is + // deduped away (returns false). The capture is BEFORE the dedup, so BOTH + // positions are recorded regardless. + const okFirst = tryEmitEdge( + graph, + indexes, + lookup, + callSite('scope:m', 11, 0), + target, + 'call', + seen, + 0.85, + true, + { sink: acc, filePath: FILE }, + ); + const okSecond = tryEmitEdge( + graph, + indexes, + lookup, + callSite('scope:m', 12, 0), + target, + 'call', + seen, + 0.85, + true, + { sink: acc, filePath: FILE }, + ); + + expect(okFirst).toBe(true); + // Collapsed dedup drops the second EDGE... + expect(okSecond).toBe(false); + expect(graph.relationships).toHaveLength(1); + // ...but BOTH call-site positions are captured. + expect(snapshot(acc, FILE)).toEqual({ + [calleeIdPosKey(11, 0)]: ['fn:target'], + [calleeIdPosKey(12, 0)]: ['fn:target'], + }); + }); + + it('R4 gating — sink undefined (pdg off): no capture and identical edge output', () => { + const callerFn = def('def:caller', 'Function', 'caller'); + const target = def('def:target', 'Function', 'target'); + const mod = scope('scope:m', null, 'Module', [callerFn, target]); + const indexes = makeIndexes([mod], [callerFn, target]); + + const withGraph = createKnowledgeGraph(); + fnNode(withGraph, 'fn:caller', 'caller'); + fnNode(withGraph, 'fn:target', 'target'); + const withLookup = buildGraphNodeLookup(withGraph); + const acc = createCalleeIdAccumulator(); + tryEmitEdge( + withGraph, + indexes, + withLookup, + callSite('scope:m', 7, 3), + target, + 'call', + new Set(), + 0.85, + false, + { sink: acc, filePath: FILE }, + ); + + const offGraph = createKnowledgeGraph(); + fnNode(offGraph, 'fn:caller', 'caller'); + fnNode(offGraph, 'fn:target', 'target'); + const offLookup = buildGraphNodeLookup(offGraph); + tryEmitEdge( + offGraph, + indexes, + offLookup, + callSite('scope:m', 7, 3), + target, + 'call', + new Set(), + 0.85, + false, + undefined, + ); + + // pdg-off: nothing captured. + expect(offGraph.relationships).toHaveLength(1); + // The EDGE rows are byte-identical between on and off (only capture differs). + expect(offGraph.relationships).toEqual(withGraph.relationships); + // The on-run DID capture (so the comparison is meaningful, not vacuous). + expect(snapshot(acc, FILE)).toEqual({ [calleeIdPosKey(7, 3)]: ['fn:target'] }); + }); +}); + +// ─── Path 1b: tryEmitEdgeWithExplicitTargetId ───────────────────────────── + +describe('callee-id capture — tryEmitEdgeWithExplicitTargetId', () => { + it('captures the explicit target id at the call-site position', () => { + const callerFn = def('def:caller', 'Function', 'caller'); + const mod = scope('scope:m', null, 'Module', [callerFn]); + const indexes = makeIndexes([mod], [callerFn]); + + const graph = createKnowledgeGraph(); + fnNode(graph, 'fn:caller', 'caller'); + const lookup = buildGraphNodeLookup(graph); + + const acc = createCalleeIdAccumulator(); + const ok = tryEmitEdgeWithExplicitTargetId( + graph, + indexes, + lookup, + callSite('scope:m', 42, 6), + 'fn:explicitTarget', + 'global', + new Set(), + 0.85, + false, + { sink: acc, filePath: FILE }, + ); + + expect(ok).toBe(true); + expect(snapshot(acc, FILE)).toEqual({ + [calleeIdPosKey(42, 6)]: ['fn:explicitTarget'], + }); + }); +}); + +// ─── Path 3: emitReferencesViaLookup ────────────────────────────────────── + +describe('callee-id capture — emitReferencesViaLookup', () => { + it('captures a CALLS emitted via the inline addRelationship', () => { + const callerFn = def('def:saveUser', 'Function', 'saveUser'); + const targetFn = def('def:User.save', 'Method', 'User.save'); + const mod = scope('scope:m', null, 'Module', [callerFn, targetFn]); + const indexes = makeIndexes([mod], [callerFn, targetFn]); + + const graph = createKnowledgeGraph(); + fnNode(graph, 'fn:saveUser', 'saveUser'); + graph.addNode({ + id: 'm:User.save', + label: 'Method' as NodeLabel, + properties: { name: 'save', filePath: FILE, qualifiedName: 'User.save' }, + }); + const lookup = buildGraphNodeLookup(graph); + + const ref: Reference = { + fromScope: 'scope:m', + toDef: 'def:User.save', + atRange: range(10, 4, 10, 8), + kind: 'call', + confidence: 0.75, + evidence: [], + }; + const referenceIndex = { + bySourceScope: new Map([['scope:m', [ref]]]), + }; + + const acc = createCalleeIdAccumulator(); + const result = emitReferencesViaLookup(graph, indexes, referenceIndex, lookup, undefined, acc); + + expect(result.emitted).toBe(1); + const targetId = graph.relationships[0]!.targetId; + expect(snapshot(acc, FILE)).toEqual({ + [calleeIdPosKey(10, 4)]: [targetId], + }); + }); + + it('R4 gating — sink undefined: identical edges, nothing captured', () => { + const callerFn = def('def:caller', 'Function', 'caller'); + const targetFn = def('def:helper', 'Function', 'helper'); + const mod = scope('scope:m', null, 'Module', [callerFn, targetFn]); + const indexes = makeIndexes([mod], [callerFn, targetFn]); + + const ref: Reference = { + fromScope: 'scope:m', + toDef: 'def:helper', + atRange: range(5, 2, 5, 8), + kind: 'call', + confidence: 0.8, + evidence: [], + }; + const referenceIndex = { + bySourceScope: new Map([['scope:m', [ref]]]), + }; + + const mkGraph = (): KnowledgeGraph => { + const g = createKnowledgeGraph(); + fnNode(g, 'fn:caller', 'caller'); + fnNode(g, 'fn:helper', 'helper'); + return g; + }; + + const onGraph = mkGraph(); + const acc = createCalleeIdAccumulator(); + emitReferencesViaLookup( + onGraph, + indexes, + referenceIndex, + buildGraphNodeLookup(onGraph), + undefined, + acc, + ); + + const offGraph = mkGraph(); + emitReferencesViaLookup( + offGraph, + indexes, + referenceIndex, + buildGraphNodeLookup(offGraph), + undefined, + undefined, + ); + + expect(offGraph.relationships).toEqual(onGraph.relationships); + expect(offGraph.relationships).toHaveLength(1); + expect(snapshot(acc, FILE)).toEqual({ + [calleeIdPosKey(5, 2)]: [onGraph.relationships[0]!.targetId], + }); + }); +}); + +// ─── Path 2: emitFreeCallFallback (regression guard for the "only tryEmitEdge" bug) ─ + +describe('callee-id capture — emitFreeCallFallback (inline addRelationship)', () => { + // Build a real (hand-constructed, fully-typed) ParsedFile whose Module-scope + // bindings resolve a free call `helper()` to a local Function — so + // emitFreeCallFallback emits a CALLS via its own inline addRelationship and + // the capture line runs. + const FREE_FILE = 'free.ts'; + const targetDef = def('def:helper', 'Function', 'helper', FREE_FILE); + const callerDef = def('def:main', 'Function', 'main', FREE_FILE); + + const freeCallSite: ReferenceSite = { + name: 'helper', + atRange: range(3, 2, 3, 8), + inScope: 'scope:free-mod', + kind: 'call', + callForm: 'free', + arity: 0, + }; + + const moduleScope = scope( + 'scope:free-mod', + null, + 'Module', + [callerDef, targetDef], + range(1, 0, 100, 0), + FREE_FILE, + { helper: [{ def: targetDef, origin: 'local' }] }, + ); + + const parsed: ParsedFile = { + filePath: FREE_FILE, + moduleScope: 'scope:free-mod', + scopes: [moduleScope], + parsedImports: [], + localDefs: [callerDef, targetDef], + referenceSites: [freeCallSite], + }; + + const buildDriver = (): { + graph: KnowledgeGraph; + indexes: ScopeResolutionIndexes; + lookup: GraphNodeLookup; + } => { + const indexes = makeIndexes([moduleScope], [callerDef, targetDef]); + const graph = createKnowledgeGraph(); + fnNode(graph, 'fn:main', 'main', FREE_FILE); + fnNode(graph, 'fn:helper', 'helper', FREE_FILE); + return { graph, indexes, lookup: buildGraphNodeLookup(graph) }; + }; + + it('captures the resolved callee id at the free-call site', () => { + const { graph, indexes, lookup } = buildDriver(); + const model = createSemanticModel(); + const workspaceIndex = buildWorkspaceResolutionIndex([parsed]); + const acc = createCalleeIdAccumulator(); + + const emitted = emitFreeCallFallback( + graph, + indexes, + [parsed], + lookup, + { bySourceScope: new Map() }, + new Set(), + model, + workspaceIndex, + { calleeIdSink: acc }, + ); + + expect(emitted).toBe(1); + const callsEdge = graph.relationships.find((r) => r.type === 'CALLS')!; + expect(callsEdge.targetId).toBe('fn:helper'); + expect(snapshot(acc, FREE_FILE)).toEqual({ + [calleeIdPosKey(3, 2)]: ['fn:helper'], + }); + }); + + it('R4 gating — sink undefined: same CALLS edge, nothing captured', () => { + const on = buildDriver(); + const off = buildDriver(); + const model = createSemanticModel(); + const workspaceIndex = buildWorkspaceResolutionIndex([parsed]); + const acc = createCalleeIdAccumulator(); + + emitFreeCallFallback( + on.graph, + on.indexes, + [parsed], + on.lookup, + { bySourceScope: new Map() }, + new Set(), + model, + workspaceIndex, + { calleeIdSink: acc }, + ); + emitFreeCallFallback( + off.graph, + off.indexes, + [parsed], + off.lookup, + { bySourceScope: new Map() }, + new Set(), + createSemanticModel(), + buildWorkspaceResolutionIndex([parsed]), + {}, + ); + + expect(off.graph.relationships).toEqual(on.graph.relationships); + expect(off.graph.relationships.filter((r) => r.type === 'CALLS')).toHaveLength(1); + expect(snapshot(acc, FREE_FILE)).toEqual({ + [calleeIdPosKey(3, 2)]: ['fn:helper'], + }); + }); +}); + +describe('callee-id accumulator — delete (R6 per-file release)', () => { + it('delete(file) frees that file map and leaves other files intact', () => { + const acc = createCalleeIdAccumulator(); + acc.add('a.ts', 2, 4, 'fn:a'); + acc.add('b.ts', 5, 0, 'fn:b'); + expect(snapshot(acc, 'a.ts')).toEqual({ [calleeIdPosKey(2, 4)]: ['fn:a'] }); + + acc.delete('a.ts'); + + expect(acc.get('a.ts')).toBeUndefined(); + expect(snapshot(acc, 'b.ts')).toEqual({ [calleeIdPosKey(5, 0)]: ['fn:b'] }); + }); + + it('delete of an absent file is a no-op', () => { + const acc = createCalleeIdAccumulator(); + acc.add('b.ts', 5, 0, 'fn:b'); + acc.delete('missing.ts'); + expect(snapshot(acc, 'b.ts')).toEqual({ [calleeIdPosKey(5, 0)]: ['fn:b'] }); + }); +}); diff --git a/gitnexus/test/unit/pdg-impact-engine.test.ts b/gitnexus/test/unit/pdg-impact-engine.test.ts new file mode 100644 index 000000000..e163df4be --- /dev/null +++ b/gitnexus/test/unit/pdg-impact-engine.test.ts @@ -0,0 +1,354 @@ +import { describe, expect, it } from 'vitest'; +import { IMPACT_MAX_DEPTH } from '../../src/mcp/tools.js'; +import { + pdgLayerStatus, + runImpactPDG, + type RunPdgImpactDeps, +} from '../../src/mcp/local/pdg-impact.js'; + +describe('runImpactPDG', () => { + it('clamps huge maxDepth values to the documented impact traversal cap', async () => { + let bfsQueries = 0; + const exec = async (_repo: string, query: string) => { + if (query.includes('MATCH (a:BasicBlock) WHERE')) { + return [{ id: 'BasicBlock:src/hot.ts:1:0:0' }]; + } + if (query.includes('MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock)')) { + bfsQueries += 1; + return [{ id: `BasicBlock:src/hot.ts:${bfsQueries + 1}:0:0` }]; + } + if (query.includes('MATCH (b:BasicBlock) WHERE b.id IN $ids')) return []; + if (query.includes('MATCH (s:`Function`)')) return []; + return []; + }; + + const result = await runImpactPDG({ + repo: { lbugPath: 'repo' }, + sym: { id: 'func:hot', name: 'hot', filePath: 'src/hot.ts', startLine: 0, endLine: 0 }, + symType: 'Function', + direction: 'downstream', + maxDepth: Number.MAX_SAFE_INTEGER, + limit: 50, + executeParameterized: exec as any, + }); + + expect(bfsQueries).toBe(IMPACT_MAX_DEPTH); + expect(result.truncated).toBe(true); + expect(result.truncatedBy).toBe('depth'); + }); + + it('keeps multiple reachable BasicBlocks on the same source line as separate statements', async () => { + let bfsQueries = 0; + const sameLineA = 'BasicBlock:src/hot.ts:1:0:1'; + const sameLineB = 'BasicBlock:src/hot.ts:1:0:2'; + const exec = async (_repo: string, query: string) => { + if (query.includes('MATCH (a:BasicBlock) WHERE')) { + return [{ id: 'BasicBlock:src/hot.ts:1:0:0' }]; + } + if (query.includes('MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock)')) { + bfsQueries += 1; + return bfsQueries === 1 ? [{ id: sameLineB }, { id: sameLineA }] : []; + } + if (query.includes('MATCH (b:BasicBlock) WHERE b.id IN $ids')) { + return [ + { id: sameLineB, line: 2, text: 'b();' }, + { id: sameLineA, line: 2, text: 'a();' }, + ]; + } + if (query.includes('MATCH (s:`Function`)')) { + return [{ id: 'func:hot', name: 'hot', label: 'Function', startLine: 0 }]; + } + return []; + }; + + const result = await runImpactPDG({ + repo: { lbugPath: 'repo' }, + sym: { id: 'func:hot', name: 'hot', filePath: 'src/hot.ts', startLine: 0, endLine: 3 }, + symType: 'Function', + direction: 'downstream', + maxDepth: 2, + limit: 50, + line: 1, + executeParameterized: exec as any, + }); + + expect(result.mode).toBe('pdg'); + expect((result as any).affectedStatementCount).toBe(2); + expect((result as any).affectedStatements.map((s: any) => s.line)).toEqual([2, 2]); + expect((result as any).affectedStatements.map((s: any) => s.text)).toEqual(['a();', 'b();']); + expect((result as any).pdgEvidence.statements).toBe('local-dependence'); + expect((result as any).pdgEvidence.localSymbols).toBe('owner-projection'); + expect((result as any).byDepth[1][0].pdgEvidence).toBe('owner-projection'); + }); + + it('tags affectedStatements scope=intra for criterion-function lines, scope=inter for cross-function reach (FU-A)', async () => { + // MIXED slice: the criterion function owns fnLine 1 in src/a.ts (sym.startLine + // 0 → ownerFnLine 1). The dependence reach surfaces TWO blocks — one in the + // criterion's own function (fnLine 1, INTRA) and one in a callee function that + // starts at line 5 (fnLine 5, reached across the call boundary → INTER). The + // scope tag is a pure parse of the block id against (criterionFile, ownerFnLine); + // each statement must carry the right tag. + const seed = 'BasicBlock:src/a.ts:1:0:0'; // criterion fn, fnLine 1 + const intraReach = 'BasicBlock:src/a.ts:1:0:1'; // same fn → INTRA + const interReach = 'BasicBlock:src/a.ts:5:0:0'; // callee fn fnLine 5 → INTER + let bfsQueries = 0; + const exec: RunPdgImpactDeps['executeParameterized'] = async (_repo, query) => { + if (query.includes('MATCH (a:BasicBlock) WHERE')) { + return [{ id: seed }]; + } + if (query.includes('MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock)')) { + bfsQueries += 1; + return bfsQueries === 1 ? [{ id: intraReach }, { id: interReach }] : []; + } + // Interproc descent calleeIds probe → no callees, so the descent is a no-op; + // the cross-function block already entered the reachable set above. + if (query.includes('RETURN b.calleeIds AS calleeIds')) return []; + if (query.includes('MATCH (b:BasicBlock) WHERE b.id IN $ids')) { + return [ + { id: intraReach, line: 2, text: 'x = local();' }, + { id: interReach, line: 6, text: 'return callee();' }, + ]; + } + if (query.includes('MATCH (s:`Function`)')) return []; + return []; + }; + + const result = await runImpactPDG({ + repo: { lbugPath: 'repo' }, + sym: { id: 'func:a', name: 'a', filePath: 'src/a.ts', startLine: 0, endLine: 3 }, + symType: 'Function', + direction: 'downstream', + maxDepth: 2, + limit: 50, + line: 1, + executeParameterized: exec, + }); + + // Narrow to a result that carries the statement slice (no `as any`). + expect('affectedStatements' in result).toBe(true); + const statements = 'affectedStatements' in result ? result.affectedStatements : []; + // Sorted by line: the intra block (line 2) precedes the inter block (line 6). + expect(statements).toMatchObject([ + { line: 2, filePath: 'src/a.ts', scope: 'intra' }, + { line: 6, filePath: 'src/a.ts', scope: 'inter' }, + ]); + }); + + it('pins the owning function: a same-source-line closure block does not leak into the seed', async () => { + // Symbol starts at 0 → owning fnLine === 1. The seed query (a forgiving + // startLine-within-window match) returns BOTH the symbol's own block at the + // seeded line AND a closure body block that happens to start on the same + // source line but is owned by a function starting at line 5 (fnLine 5). + const owned = 'BasicBlock:src/hot.ts:1:0:3'; // fnLine 1 === sym.startLine + 1 + const closureLeak = 'BasicBlock:src/hot.ts:5:10:0'; // fnLine 5 — a nested closure + const exec: RunPdgImpactDeps['executeParameterized'] = async (_repo, query) => { + if (query.includes('MATCH (a:BasicBlock) WHERE')) { + return [{ id: owned }, { id: closureLeak }]; + } + // No downstream reachability — exercises the seedBlocks-carrying branch. + if (query.includes('MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock)')) return []; + if (query.includes('MATCH (b:BasicBlock) WHERE b.id IN $ids')) return []; + if (query.includes('MATCH (s:`Function`)')) return []; + return []; + }; + + const result = await runImpactPDG({ + repo: { lbugPath: 'repo' }, + sym: { id: 'func:hot', name: 'hot', filePath: 'src/hot.ts', startLine: 0, endLine: 20 }, + symType: 'Function', + direction: 'downstream', + maxDepth: 2, + limit: 50, + line: 7, + executeParameterized: exec, + }); + + // Only the owning-function block survives; the closure block is dropped. + // Unconditional match: fails if `seedBlocks` is absent, has the wrong + // length, or contains the closure block — no vacuous branch. + expect(result).toMatchObject({ seedBlocks: [owned] }); + }); + + it('U-C4: ascends a return-flowing callee result into the coalesced caller call block (interior lines surface)', async () => { + // The inter-pipeline-stages shape: the criterion fn (fnLine 1) seeds at line + // 2; its dependence reaches ONE coalesced call block spanning source lines + // 4-6 (`acc = stage(acc)` ×3) that invokes a callee with a return-flow + // CALL_SUMMARY (`r:1` ⇒ formal[0] → return). Without ascent the block projects + // to its startLine (4) only; the ascent surfaces interior lines 5 and 6. + const seed = 'BasicBlock:src/p.ts:1:0:0'; // criterion fn, fnLine 1, seeded line 2 + const callBlock = 'BasicBlock:src/p.ts:1:0:5'; // coalesced 3-call block, lines 4-6 + let bfsCalls = 0; + const exec: RunPdgImpactDeps['executeParameterized'] = async (_repo, query, _params) => { + // Seed anchor → the criterion's seed block. + if (query.includes('MATCH (a:BasicBlock) WHERE')) return [{ id: seed }]; + // Intra BFS: hop 1 reaches the call block; subsequent hops (incl. the + // ascent re-seed FROM the call block) find nothing new. + if (query.includes('MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock)')) { + bfsCalls += 1; + return bfsCalls === 1 ? [{ id: callBlock }] : []; + } + // Per-block calleeIds (descent's block→callee map): the call block invokes + // one resolved callee. + if (query.includes('RETURN b.id AS id, b.calleeIds AS calleeIds')) { + return [{ id: callBlock, calleeIds: 'Function:src/p.ts:stage' }]; + } + // CALL_SUMMARY self-loop: stage has a non-empty return-flow (`r:1`). + if (query.includes("r.type = 'CALL_SUMMARY'")) { + return [{ id: 'Function:src/p.ts:stage', reason: '1|r:1' }]; + } + // FU-B-2 self REACHING_DEF edge of the ascent call block: the coalesced + // same-binding `acc` reassignment chain (def 4→use 5, def 5→use 6) is ONE + // deduped edge whose `reason` carries the FULL ordered pair LIST + // (`acc|1:4:5;5:6`). The interior-line walk follows the whole list to + // fixpoint, surfacing both line 5 and line 6 from the block start (4) — a + // first-pair-only encoding would surface only line 5. + if (query.includes('MATCH (a:BasicBlock)-[r:CodeRelation]->(a)')) { + return [{ id: callBlock, reason: 'acc|1:4:5;5:6' }]; + } + // Callee span resolution: stage has no CFG body here (no callee blocks to + // descend into — the ascent, not the descent, is under test). + if (query.includes('MATCH (s:`Function`)')) return []; + // Statement projection over the reachable set (the call block only — the + // seed is excluded by the seed-minus-reachable convention). + if (query.includes('RETURN b.id AS id, b.startLine AS line, b.endLine AS endLine')) { + return [ + { + id: callBlock, + line: 4, + endLine: 6, + text: 'acc = stageA(acc);\nacc = stageB(acc);\nacc = stageC(acc);', + }, + ]; + } + return []; + }; + + const result = await runImpactPDG({ + repo: { lbugPath: 'repo' }, + sym: { + id: 'Function:src/p.ts:run', + name: 'run', + filePath: 'src/p.ts', + startLine: 0, + endLine: 7, + }, + symType: 'Function', + direction: 'downstream', + maxDepth: 3, + limit: 50, + line: 2, + executeParameterized: exec, + callSummaryAvailable: true, + }); + + expect('affectedStatements' in result).toBe(true); + const statements = 'affectedStatements' in result ? result.affectedStatements : []; + // The coalesced call block expands to ALL THREE interior lines (4,5,6) with + // their own text — the ascent win. Without U-C4 only line 4 would appear. + expect(statements).toMatchObject([ + { line: 4, filePath: 'src/p.ts', scope: 'intra', text: 'acc = stageA(acc);' }, + { line: 5, filePath: 'src/p.ts', scope: 'intra', text: 'acc = stageB(acc);' }, + { line: 6, filePath: 'src/p.ts', scope: 'intra', text: 'acc = stageC(acc);' }, + ]); + }); + + it('U-C4: an EMPTY (r:0) call summary does NOT ascend — the call block stays single-line (sound default)', async () => { + // Same shape, but stage's CALL_SUMMARY records NO return-flow (`r:0`): the + // call result does NOT depend on the slice, so the coalesced block must NOT + // expand — it projects to its startLine only (no false ascent). + const seed = 'BasicBlock:src/p.ts:1:0:0'; + const callBlock = 'BasicBlock:src/p.ts:1:0:5'; + let bfsCalls = 0; + const exec: RunPdgImpactDeps['executeParameterized'] = async (_repo, query) => { + if (query.includes('MATCH (a:BasicBlock) WHERE')) return [{ id: seed }]; + if (query.includes('MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock)')) { + bfsCalls += 1; + return bfsCalls === 1 ? [{ id: callBlock }] : []; + } + if (query.includes('RETURN b.id AS id, b.calleeIds AS calleeIds')) { + return [{ id: callBlock, calleeIds: 'Function:src/p.ts:stage' }]; + } + // Empty return-flow → calleesWithReturnFlow yields NO entry. + if (query.includes("r.type = 'CALL_SUMMARY'")) { + return [{ id: 'Function:src/p.ts:stage', reason: '1|r:0' }]; + } + if (query.includes('MATCH (s:`Function`)')) return []; + if (query.includes('RETURN b.id AS id, b.startLine AS line, b.endLine AS endLine')) { + return [ + { id: callBlock, line: 4, endLine: 6, text: 'acc = stageA(acc);\nacc = stageB(acc);' }, + ]; + } + return []; + }; + + const result = await runImpactPDG({ + repo: { lbugPath: 'repo' }, + sym: { + id: 'Function:src/p.ts:run', + name: 'run', + filePath: 'src/p.ts', + startLine: 0, + endLine: 7, + }, + symType: 'Function', + direction: 'downstream', + maxDepth: 3, + limit: 50, + line: 2, + executeParameterized: exec, + callSummaryAvailable: true, + }); + + const statements = 'affectedStatements' in result ? result.affectedStatements : []; + // Only the block's startLine (4) — no interior expansion (empty summary). + expect(statements).toMatchObject([{ line: 4, filePath: 'src/p.ts', scope: 'intra' }]); + expect(statements.map((s) => s.line)).toEqual([4]); + }); +}); + +describe('pdgLayerStatus', () => { + const unreadableMeta = async () => null as any; + + it('reports visible PDG edges as unknown without a probe error when meta is unreadable', async () => { + const result = await pdgLayerStatus({ + lbugPath: 'repo/.gitnexus/lbug', + loadMetaFn: unreadableMeta, + executeParameterized: (async (_repo: string, query: string) => { + expect(query).toContain('LIMIT 1'); + return [{ type: 'CDG' }]; + }) as any, + }); + + expect(result.state).toBe('unknown'); + expect(result.note).toContain('edges ARE visible'); + expect(result.probeError).toBeUndefined(); + }); + + it('reports no visible PDG edges separately from probe failures', async () => { + const result = await pdgLayerStatus({ + lbugPath: 'repo/.gitnexus/lbug', + loadMetaFn: unreadableMeta, + executeParameterized: (async () => []) as any, + }); + + expect(result.state).toBe('unknown'); + expect(result.note).toContain('no CDG/REACHING_DEF edges visible'); + expect(result.probeError).toBeUndefined(); + }); + + it('preserves probe failures instead of reporting a false no-edge signal', async () => { + const result = await pdgLayerStatus({ + lbugPath: 'repo/.gitnexus/lbug', + loadMetaFn: unreadableMeta, + executeParameterized: (async () => { + throw new Error('database busy'); + }) as any, + }); + + expect(result.state).toBe('unknown'); + expect(result.probeError).toBe('database busy'); + expect(result.note).toContain('probe failed'); + expect(result.note).not.toContain('no CDG/REACHING_DEF edges visible'); + expect(result.recoverySuggestion).toContain('LadybugDB'); + }); +}); diff --git a/gitnexus/test/unit/pdg-mode-flip.test.ts b/gitnexus/test/unit/pdg-mode-flip.test.ts index 585747eb7..4b5092532 100644 --- a/gitnexus/test/unit/pdg-mode-flip.test.ts +++ b/gitnexus/test/unit/pdg-mode-flip.test.ts @@ -295,6 +295,7 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxInterprocEdges: 1000, taintModelVersion, reachingDefSolver: 'ssa-sparse-v1', + hasCallSummary: true, }); expect(stamped!.incrementalInProgress).toBeUndefined(); // cleared on success @@ -352,6 +353,7 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxInterprocEdges: 1000, taintModelVersion, reachingDefSolver: 'ssa-sparse-v1', + hasCallSummary: true, }); // The CFG layer survives a rebuild under a tighter edge cap (blocks are // never capped, only edges). diff --git a/gitnexus/test/unit/run-analyze.test.ts b/gitnexus/test/unit/run-analyze.test.ts index b317d6824..791cda8ae 100644 --- a/gitnexus/test/unit/run-analyze.test.ts +++ b/gitnexus/test/unit/run-analyze.test.ts @@ -353,6 +353,9 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { // (#2201 review R3). Bumps when the reaching-defs solver's emitted facts // change; absence on a pre-#2201 stamp forces a re-analysis. reachingDefSolver: 'ssa-sparse-v1', + // FU-C return-value-ascent layer presence — always stamped on a pdg-on run; + // absence on a pre-FU-C (v3) stamp forces a re-analysis (key-union mismatch). + hasCallSummary: true, }; it('resolvePdgConfig: pdg-off run resolves to undefined (the meta field is omitted)', async () => { @@ -389,6 +392,7 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { maxInterprocEdges: 0, taintModelVersion, // not a cap — always stamped on a pdg-on run reachingDefSolver: 'ssa-sparse-v1', // solver identity — always stamped (#2201 R3) + hasCallSummary: true, // FU-C ascent layer — always stamped on a pdg-on run }); }); diff --git a/gitnexus/test/unit/security.test.ts b/gitnexus/test/unit/security.test.ts index 37cb9f6f8..906ae305e 100644 --- a/gitnexus/test/unit/security.test.ts +++ b/gitnexus/test/unit/security.test.ts @@ -73,6 +73,10 @@ describe('VALID_RELATION_TYPES', () => { // types" sweep can't drag them in, mirroring the TAINTED/TAINT_PATH pins. expect(VALID_RELATION_TYPES.has('CDG')).toBe(false); expect(VALID_RELATION_TYPES.has('POST_DOMINATE')).toBe(false); + // REACHING_DEF is the other BasicBlock→BasicBlock PDG edge (#2086 impact + // PDG mode traverses it directly, never through impact's symbol-space BFS). + // Pinned alongside CDG/POST_DOMINATE so an allow-all sweep can't drag it in. + expect(VALID_RELATION_TYPES.has('REACHING_DEF')).toBe(false); }); }); diff --git a/gitnexus/test/unit/taint/call-summary-codec.test.ts b/gitnexus/test/unit/taint/call-summary-codec.test.ts new file mode 100644 index 000000000..30fe88f61 --- /dev/null +++ b/gitnexus/test/unit/taint/call-summary-codec.test.ts @@ -0,0 +1,115 @@ +/** + * PDG FU-C (U-C2) — the shared CALL_SUMMARY reason bitset codec. + * + * The wire format must round-trip BYTE-EXACT through the CSV persistence layer + * (`escapeCSVField ∘ sanitizeUTF8`, csv-generator.ts) — that composition is + * exercised here verbatim, including the un-escape a DB load performs. Malformed + * input must produce a typed failure, never a throw; an unknown trailing segment + * (a future out-param/exception fact) must decode without error (forward compat). + */ + +import { describe, it, expect } from 'vitest'; +import { + CALL_SUMMARY_CODEC_VERSION, + encodeCallSummary, + decodeCallSummary, +} from '../../../src/core/ingestion/taint/call-summary-codec.js'; +import { escapeCSVField, sanitizeUTF8 } from '../../../src/core/lbug/csv-generator.js'; + +/** Inverse of escapeCSVField — what a CSV/DB load applies to the stored cell. */ +function unescapeCSVField(cell: string): string { + expect(cell.startsWith('"') && cell.endsWith('"')).toBe(true); + return cell.slice(1, -1).replace(/""/g, '"'); +} + +const roundTrip = (params: readonly number[]) => { + const reason = encodeCallSummary(params); + const decoded = decodeCallSummary(reason); + expect(decoded).toMatchObject({ ok: true }); + return { reason, decoded }; +}; + +describe('encodeCallSummary / decodeCallSummary round trip', () => { + it('round-trips a single return-flowing formal index (formal 0 flows, 1 does not)', () => { + const { decoded } = roundTrip([0]); + expect(decoded).toMatchObject({ + ok: true, + version: CALL_SUMMARY_CODEC_VERSION, + returnFlowParams: [0], + }); + }); + + it('round-trips multiple non-contiguous return-flowing indices, sorted + deduped', () => { + const { decoded } = roundTrip([5, 0, 5, 2]); + expect(decoded).toMatchObject({ ok: true, returnFlowParams: [0, 2, 5] }); + }); + + it('encodes an empty ascent as an explicit empty summary (r:0)', () => { + const reason = encodeCallSummary([]); + expect(reason).toBe(`${CALL_SUMMARY_CODEC_VERSION}|r:0`); + expect(decodeCallSummary(reason)).toMatchObject({ ok: true, returnFlowParams: [] }); + }); + + it('handles a high formal index beyond 32 bits (BigInt bitset, no overflow)', () => { + const { decoded } = roundTrip([64]); + expect(decoded).toMatchObject({ ok: true, returnFlowParams: [64] }); + }); + + it('survives the CSV escape/sanitize/unescape persistence path byte-exact', () => { + const reason = encodeCallSummary([0, 3, 7]); + const persisted = escapeCSVField(sanitizeUTF8(reason)); + const loaded = unescapeCSVField(persisted); + expect(loaded).toBe(reason); + expect(decodeCallSummary(loaded)).toMatchObject({ ok: true, returnFlowParams: [0, 3, 7] }); + }); + + it('ignores negative / non-integer indices on encode (defensive)', () => { + const { decoded } = roundTrip([1, -1, 2.5, 3]); + expect(decoded).toMatchObject({ ok: true, returnFlowParams: [1, 3] }); + }); +}); + +describe('decodeCallSummary forward compatibility + typed failures', () => { + it('accepts and ignores an unknown trailing segment (reserved future fact)', () => { + const reason = `${CALL_SUMMARY_CODEC_VERSION}|r:5|o:deadbeef`; + expect(decodeCallSummary(reason)).toMatchObject({ ok: true, returnFlowParams: [0, 2] }); + }); + + it('rejects an unsupported version with a typed failure (no throw)', () => { + expect(decodeCallSummary('9|r:1')).toMatchObject({ ok: false }); + }); + + it('rejects a MULTI-DIGIT future version cleanly (parsed before the first `|`, not as one char)', () => { + // A future writer emitting `12|…` must degrade to a clean unsupported-version + // typed-failure — NOT mis-parse the version as '1' with a stray '2' segment. + // The error names the FULL multi-digit token, proving it was not truncated to + // a single char. + expect(decodeCallSummary('12|r:1')).toMatchObject({ + ok: false, + error: "unsupported call-summary version '12'", + }); + // A 3-digit version likewise fails as a whole token (no `>9` truncation). + expect(decodeCallSummary('123|r:5')).toMatchObject({ + ok: false, + error: "unsupported call-summary version '123'", + }); + }); + + it('rejects an empty / non-string reason', () => { + expect(decodeCallSummary('')).toMatchObject({ ok: false }); + expect(decodeCallSummary(undefined)).toMatchObject({ ok: false }); + expect(decodeCallSummary(42)).toMatchObject({ ok: false }); + }); + + it('rejects a missing return segment', () => { + expect(decodeCallSummary(`${CALL_SUMMARY_CODEC_VERSION}|o:ff`)).toMatchObject({ ok: false }); + }); + + it('rejects a non-hex return payload', () => { + expect(decodeCallSummary(`${CALL_SUMMARY_CODEC_VERSION}|r:xyz`)).toMatchObject({ ok: false }); + }); + + it('rejects a body without the segment separator', () => { + expect(decodeCallSummary(`${CALL_SUMMARY_CODEC_VERSION}r:1`)).toMatchObject({ ok: false }); + }); +}); diff --git a/gitnexus/test/unit/taint/call-summary-harvest.test.ts b/gitnexus/test/unit/taint/call-summary-harvest.test.ts new file mode 100644 index 000000000..fa21160d7 --- /dev/null +++ b/gitnexus/test/unit/taint/call-summary-harvest.test.ts @@ -0,0 +1,122 @@ +/** + * PDG FU-C (U-C2) — per-function RETURN-VALUE ASCENT harvest soundness. + * + * Fixtures parse REAL TypeScript through the shared CFG harness, so the harvester + * consumes the exact `FunctionCfg` / `FunctionDefUse` the pipeline produces. The + * load-bearing invariant proved here: `returnFlowParams` is either the correct + * 0-based ENCLOSING FORMAL positions or empty — NEVER a flattened binding ordinal + * that misattributes a destructured/rest formal's flow to a later simple formal + * (the consumer reads the bitset positionally, so an ordinal would be UNSOUND). + */ + +import { describe, it, expect } from 'vitest'; +import { cfgOf } from '../../helpers/ts-cfg-harness.js'; +import type { FunctionCfg } from '../../../src/core/ingestion/cfg/types.js'; +import { computeReachingDefs } from '../../../src/core/ingestion/cfg/reaching-defs.js'; +import { harvestCallSummary } from '../../../src/core/ingestion/taint/call-summary-harvest.js'; + +function harvest(code: string, fnIndex = 0) { + const cfg: FunctionCfg = cfgOf(code, fnIndex); + const defUse = computeReachingDefs(cfg); + return harvestCallSummary(cfg, defUse); +} + +describe('harvestCallSummary — simple-formal precision', () => { + it('records a single param flowing straight to the return as formal 0', () => { + expect(harvest(`function f(x: string) { return x; }`)).toMatchObject({ + status: 'computed', + facts: { paramCount: 1, returnFlowParams: [0] }, + }); + }); + + it('records the SECOND simple formal flowing to return as formal 1 (not 0)', () => { + expect(harvest(`function f(a: string, b: string) { return b; }`)).toMatchObject({ + status: 'computed', + facts: { paramCount: 2, returnFlowParams: [1] }, + }); + }); + + it('records a param returned through a local assignment', () => { + expect(harvest(`function f(x: string) { const y = x; return y; }`).facts).toMatchObject({ + returnFlowParams: [0], + }); + }); + + it('records only the flowing formal among several simple params', () => { + expect(harvest(`function add(a: number, b: number) { return a + 1; }`).facts).toMatchObject({ + paramCount: 2, + returnFlowParams: [0], + }); + }); +}); + +describe('harvestCallSummary — destructured / rest formal SOUNDNESS (no false attribution)', () => { + it('destructured formal BEFORE a simple formal yields the formal-0 position, NEVER ordinal 1', () => { + // `function f({a, b}, c) { return b }` — b is an inner name of the destructured + // formal at position 0, NOT formal 1 (= c). The flattened binding ordinal of b + // is 1, which the positional consumer would read as "c flows" — a FALSE claim. + // The fix keys on the enclosing FORMAL position, so this is [0], and crucially + // it is NEVER [1]. + const facts = harvest( + `function f({ a, b }: { a: string; b: string }, c: string) { return b; }`, + ).facts; + expect(facts).toMatchObject({ returnFlowParams: [0] }); + expect(facts.returnFlowParams).not.toContain(1); + }); + + it('simple formal BEFORE a destructured formal yields formal-1 for the inner name', () => { + // `function g(a, {b, c}) { return c }` — c is an inner name of the destructured + // formal at position 1, so the recorded return-flow position is 1. + expect( + harvest(`function g(a: string, { b, c }: { b: string; c: string }) { return c; }`).facts, + ).toMatchObject({ returnFlowParams: [1] }); + }); + + it('a returned simple formal that follows a rest formal keeps its own formal position', () => { + // `function r(a, ...rest) { return a }` — a is formal 0, rest is formal 1. + expect(harvest(`function r(a: string, ...rest: string[]) { return a; }`).facts).toMatchObject({ + returnFlowParams: [0], + }); + }); +}); + +describe('harvestCallSummary — conservative fallback (formalIndex absent ⇒ EMPTY, never wrong)', () => { + it('emits an EMPTY summary when a param binding lacks a producer-supplied formalIndex', () => { + // A producer that does not stamp `formalIndex` (e.g. a stale warm-cache shape) + // cannot prove ordinal == formal slot, so the harvest must fall back to EMPTY — + // a documented MISS, never a flattened ordinal. Strip the field to simulate it. + const cfg = cfgOf(`function f(a: string, b: string) { return b; }`); + const stripped: FunctionCfg = { + ...cfg, + // Re-build each binding WITHOUT formalIndex (rest-destructure drops it). + bindings: cfg.bindings?.map(({ formalIndex: _omit, ...rest }) => rest), + }; + const defUse = computeReachingDefs(stripped); + expect(harvestCallSummary(stripped, defUse)).toMatchObject({ + status: 'computed', + facts: { paramCount: 2, returnFlowParams: [] }, + }); + }); + + it('reports a coverage gap when reaching-defs is not computed (no bindings)', () => { + const cfg = cfgOf(`function f(x: string) { return x; }`); + const bare: FunctionCfg = { ...cfg, bindings: undefined }; + const defUse = computeReachingDefs(bare); + expect(harvestCallSummary(bare, defUse)).toMatchObject({ status: 'coverage-gap' }); + }); +}); + +describe('harvestCallSummary — empty/void cases', () => { + it('a void function (no return value) records no return-flow', () => { + expect(harvest(`function f(x: string) { x.trim(); }`).facts).toMatchObject({ + returnFlowParams: [], + }); + }); + + it('a param-less function records no return-flow', () => { + expect(harvest(`function f() { const a = 1; return a; }`).facts).toMatchObject({ + paramCount: 0, + returnFlowParams: [], + }); + }); +}); diff --git a/gitnexus/test/unit/taint/call-summary-language-and-constructor.test.ts b/gitnexus/test/unit/taint/call-summary-language-and-constructor.test.ts new file mode 100644 index 000000000..1749db121 --- /dev/null +++ b/gitnexus/test/unit/taint/call-summary-language-and-constructor.test.ts @@ -0,0 +1,56 @@ +// U10 — characterize two sound-but-silent CALL_SUMMARY (return-value ascent) +// coverage gaps: +// 1. Non-TS/JS languages produce EMPTY summaries — only the TS/JS harvester +// stamps the producer `formalIndex` the ascent needs. +// 2. Constructors are excluded from the functionish node index, so a +// constructor never receives a CALL_SUMMARY edge (FUNCTIONISH_LABELS). + +import { describe, it, expect } from 'vitest'; +import { cfgOf } from '../../helpers/ts-cfg-harness.js'; +import type { FunctionCfg } from '../../../src/core/ingestion/cfg/types.js'; +import { computeReachingDefs } from '../../../src/core/ingestion/cfg/reaching-defs.js'; +import { harvestCallSummary } from '../../../src/core/ingestion/taint/call-summary-harvest.js'; +import { buildFunctionNodeIndex } from '../../../src/core/ingestion/taint/summary-harvest-driver.js'; +import { createKnowledgeGraph } from '../../../src/core/graph/graph.js'; + +describe('CALL_SUMMARY language coverage (U10)', () => { + it('a non-TS/JS function (param bindings without producer formalIndex) yields an EMPTY summary', () => { + // Only the TS/JS harvester stamps `formalIndex`; every other language leaves + // it undefined, so return-value ascent is structurally empty there. Model + // that by stripping formalIndex from a real CFG's param bindings. + const cfg: FunctionCfg = cfgOf(`function f(a: string, b: string) { return b; }`); + const nonTs: FunctionCfg = { + ...cfg, + bindings: cfg.bindings?.map(({ formalIndex: _omit, ...rest }) => rest), + }; + const result = harvestCallSummary(nonTs, computeReachingDefs(nonTs)); + expect(result).toMatchObject({ status: 'computed', facts: { returnFlowParams: [] } }); + }); +}); + +describe('buildFunctionNodeIndex — Constructor exclusion (U10)', () => { + it('indexes Function and Method nodes but NOT Constructor (no return-value ascent)', () => { + const graph = createKnowledgeGraph(); + graph.addNode({ + id: 'Function:f.ts:fn', + label: 'Function', + properties: { name: 'fn', filePath: 'f.ts', startLine: 10 }, + }); + graph.addNode({ + id: 'Method:f.ts:m', + label: 'Method', + properties: { name: 'm', filePath: 'f.ts', startLine: 30 }, + }); + graph.addNode({ + id: 'Constructor:f.ts:ctor', + label: 'Constructor', + properties: { name: 'ctor', filePath: 'f.ts', startLine: 20 }, + }); + const index = buildFunctionNodeIndex(graph); + expect(index.get('f.ts')?.get(10)).toEqual(['Function:f.ts:fn']); + expect(index.get('f.ts')?.get(30)).toEqual(['Method:f.ts:m']); + // The Constructor's start line is absent → resolveFnId returns undefined for + // it (unresolved), so no CALL_SUMMARY summary is harvested. + expect(index.get('f.ts')?.get(20)).toBeUndefined(); + }); +}); diff --git a/gitnexus/test/unit/tools.test.ts b/gitnexus/test/unit/tools.test.ts index f1b3dc83b..1d8906131 100644 --- a/gitnexus/test/unit/tools.test.ts +++ b/gitnexus/test/unit/tools.test.ts @@ -134,6 +134,24 @@ describe('GITNEXUS_TOOLS', () => { expect(impactTool.inputSchema.required).toContain('direction'); }); + it('impact tool advertises the PDG-only `line` statement anchor (integer, min 1, not required)', () => { + const impactTool = GITNEXUS_TOOLS.find((t) => t.name === 'impact')!; + const line = (impactTool.inputSchema.properties as Record).line; + expect(line).toBeDefined(); + expect(line.type).toBe('integer'); + expect(line.minimum).toBe(1); + // Statement-anchored slice is optional — never required. + expect(impactTool.inputSchema.required).not.toContain('line'); + // The description names the mode:'pdg' statement-anchor semantics. + expect(line.description).toMatch(/statement anchor/i); + expect(line.description).toMatch(/pdg/i); + // The top-level description mentions the statement-anchored slice and result shape. + expect(impactTool.description).toMatch(/statement-anchored|STATEMENT-ANCHORED/); + expect(impactTool.description).toContain('affectedStatements'); + expect(impactTool.description).toContain('target metadata'); + expect(impactTool.description).toContain('truncatedBy'); + }); + it('rename tool requires new_name', () => { const renameTool = GITNEXUS_TOOLS.find((t) => t.name === 'rename')!; expect(renameTool.inputSchema.required).toContain('new_name'); @@ -276,6 +294,26 @@ describe('GITNEXUS_TOOLS', () => { expect(relProp.items).toEqual({ type: 'string' }); }); + it('impact advertises a mode param (callgraph default; pdg opt-in) — not a new tool (KTD1)', () => { + // KTD1: pdg impact ships as a PARAM on the existing tool, so the tool count + // must NOT change (asserted at 17 above) and `impact` must expose `mode`. + const impactTool = GITNEXUS_TOOLS.find((t) => t.name === 'impact')!; + const modeProp = impactTool.inputSchema.properties.mode; + expect(modeProp).toBeDefined(); + expect(modeProp.type).toBe('string'); + expect(modeProp.enum).toEqual(['callgraph', 'pdg']); + expect(modeProp.default).toBe('callgraph'); + // The description must teach the opt-in / intra-procedural / --pdg contract. + expect(modeProp.description).toContain('pdg'); + expect(modeProp.description).toContain('--pdg'); + expect(modeProp.description.toLowerCase()).toContain('intra-procedural'); + expect(modeProp.description).toContain('affectedStatements'); + expect(modeProp.description).toContain('UNKNOWN-risk'); + // The tool-level description must mention the mode so an LLM discovers it. + expect(impactTool.description.toLowerCase()).toContain('mode'); + expect(impactTool.description).toContain('pdg'); + }); + it('route_map description defers to api_impact for pre-change analysis', () => { const routeMapTool = GITNEXUS_TOOLS.find((t) => t.name === 'route_map')!; expect(routeMapTool.description).toContain('api_impact');