From 306f0ee071d47db92a8145e50f81dcbe63fcd49a Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 16 Jun 2026 09:08:10 +0000 Subject: [PATCH] feat(impact-pdg): accuracy measurement harness (U7) measure.mjs drives both impact modes over the U6 corpus via a mock-free real-analyze substrate (temp GITNEXUS_HOME + child-process analyze --pdg + LocalBackend), validates fixtures in Step 0 (>=1 PDG edge + R4 same-line guard), and computes per-mode/per-scope precision/recall/F1 + Jaccard + true/noise set-diffs (Arnold-Bohner CIS/AIS). Two-gate --check (one-sided F1 band + annotation fingerprint, median-of-K for substrate noise). Pure scorer in metrics.mjs; 18 synthetic-set metric-math unit tests stay out of the flaky pipeline lane. Key measured finding (the deliverable verdict): at SYMBOL granularity PDG mode's intra blast radius is empty (intra-procedural dependence collapses onto the criterion's own symbol), so callgraph wins for symbol-level impact; PDG v1's value is block-level dependence detail. Promotion gated on a deferred Function->BasicBlock CONTAINS_BLOCK edge. Refs U7 --- gitnexus/bench/impact-pdg/README.md | 224 +++++- gitnexus/bench/impact-pdg/baselines.json | 21 + .../inter-pipeline-stages/ground-truth.json | 4 +- gitnexus/bench/impact-pdg/measure.mjs | 639 ++++++++++++++++++ gitnexus/bench/impact-pdg/metrics.mjs | 238 +++++++ .../test/unit/impact-pdg-metric-math.test.ts | 263 +++++++ 6 files changed, 1367 insertions(+), 22 deletions(-) create mode 100644 gitnexus/bench/impact-pdg/baselines.json create mode 100644 gitnexus/bench/impact-pdg/measure.mjs create mode 100644 gitnexus/bench/impact-pdg/metrics.mjs create mode 100644 gitnexus/test/unit/impact-pdg-metric-math.test.ts diff --git a/gitnexus/bench/impact-pdg/README.md b/gitnexus/bench/impact-pdg/README.md index 656af0a42..8aaa19ba2 100644 --- a/gitnexus/bench/impact-pdg/README.md +++ b/gitnexus/bench/impact-pdg/README.md @@ -1,10 +1,11 @@ # `bench/impact-pdg` — PDG-vs-call-graph impact accuracy harness -> **STATUS: STUB (U6).** This directory currently holds only the **curated -> ground-truth fixture corpus** (U6). The measurement harness (`measure.mjs`, -> `baselines.json`) and the full methodology write-up land in **U7**. This -> README documents the annotation schema and the validity posture so the -> fixtures are reviewable on their own. +> **STATUS: LIVE (U7).** This directory holds the curated ground-truth fixture +> corpus (U6) **and** the measurement harness (`measure.mjs`, `metrics.mjs`, +> `baselines.json`). Run it with `node --import tsx bench/impact-pdg/measure.mjs` +> (build `dist/` first — see *How to run*). The harness drives both `impact` +> engines over the fixtures, prints a stratified P/R/F1 table + a plain-language +> decision recommendation, and gates regressions with `--check`. ## What this measures @@ -33,7 +34,7 @@ mature CFG/PDG support in this codebase. | `intra-control-loop` | intra | nested loop+if controllers of a stmt (upstream, CDG-reverse) | | `inter-dispatcher-thin` | inter | branch router → 3 handlers (PDG ≈ ∅ by design) | | `inter-facade-delegate` | inter | guarded sequential delegation chain | -| `inter-pipeline-stages` | inter | straight pipeline driver (upstream) | +| `inter-pipeline-stages` | inter | straight pipeline driver (downstream → 3 stages) | | `mixed-validate-then-call` | mixed | guard-dominated intra dependence + 1 callee | | `mixed-compute-and-emit` | mixed | data-flow-dominated intra dependence + 1 callee | | `mixed-guarded-dispatch` | mixed | control+data intra dependence + 2 callees | @@ -80,25 +81,208 @@ line within the criterion function; an inter entry is a different symbol). annotation's**. Call-graph gets no such home-field annotation, so the comparison is not rigged toward PDG. -## Annotation fingerprint (KTD10) +## Methodology — CIS / AIS, stratified (KTD9, Arnold–Bohner) -U7 computes an **order-independent fingerprint over this annotation set** so an -*unreviewed edit to ground truth* — which silently moves the metric — trips a -distinct `--check` gate (separate from the one-sided F1 regression band). The -canonicalizer is annotation-set-shaped (it mirrors the `bench/cfg/measure.mjs` -technique, not a literal import). +For each fixture × mode the harness compares the mode's **CIS** (Computed Impact +Set — the symbols it reports as impacted) against the **AIS** (Actual Impact Set +— the curated ground truth), stratified by impact locus: + +- **precision** = |AIS∩CIS| / |CIS| (over-approximation cost), +- **recall** = |AIS∩CIS| / |AIS| (under-approximation; the *dangerous* miss for + a safety tool), +- **F1** = harmonic mean, +- **FPIS** = CIS − AIS (noise), **FNIS** = AIS − CIS (missed), +- **|CIS|/|AIS|** size ratio, +- cross-mode **Jaccard(callgraph_CIS, pdg_CIS)** + directional set-diffs + (`pdg-only` / `callgraph-only`), each split into *true* (∩AIS) vs *noise* + (−AIS). + +**Empty-denominator semantics are explicit, never silently 0/1.** |CIS|=0 ⇒ +precision is `n/a` (no predictions); |AIS|=0 ⇒ recall is `n/a` (no truth in that +scope). A scope with an `n/a` metric is **excluded** from that metric's mean, +never folded in as 0 — folding it as 0 would punish a mode for a scope that +simply has no ground truth (the apples-to-oranges trap, R1). The pure scorer +lives in `metrics.mjs`; its arithmetic is pinned by the deterministic unit test +`test/unit/impact-pdg-metric-math.test.ts` (synthetic sets only — no analyze, no +DB, so it stays out of the flaky full-pipeline lane). + +**Granularity: symbol, never block-id.** A symbol key is `@`, +order-independent and line-collapsed. An `intra_AIS` statement-line collapses +onto its **owning symbol**; an `inter_AIS` entry already names a whole symbol. +This is why per-fixture intra-AIS reduces to the singleton `{criterion}`. CIS is +partitioned the same way: the criterion symbol itself = **intra** scope, every +other reported symbol = **inter** scope, the union = **mixed**. + +**PDG on inter-scope AIS is known-zero-recall BY DESIGN** — a capability fact, +not a loss. v1 PDG impact is intra-procedural; it cannot reach across function +boundaries. + +## Substrate (the load-bearing mechanism — R8) + +`runPipelineFromRepo` is in-memory and never persists, but `impact` queries a +**persisted** `lbugPath` + a `meta.pdg` stamp; there is no exported `runAnalyze` +(the entrypoint `analyzeCommand` calls `process.exit`, unusable in a loop), and +the test-suite `vi.mock` bridge is vitest-only. So the harness runs **real +analyze via a temp `GITNEXUS_HOME`, mock-free**. Per fixture: + +1. Point `process.env.GITNEXUS_HOME` at a per-run temp dir (honored by + `repo-manager.getGlobalDir()` — it roots the registry; the per-repo DB lands + in `/.gitnexus/`, so fixtures are copied to a temp working dir + first, keeping the source tree clean). +2. **Shell out** to the real CLI as a child process — child-process isolation + sidesteps `process.exit`; real `saveMeta` + `registerRepo` land in the temp + home; parse workers spawn from `dist/` (so the harness needs a built `dist/`): + + ``` + node --import tsx src/cli/index.ts analyze --pdg --skip-git --index-only + ``` +3. `new LocalBackend(); await init()` resolves the fixture via the **real** + registry (the parent process sets `GITNEXUS_HOME` too, so `init()` reads the + temp registry, not `~/.gitnexus`). +4. `callTool('impact', {repo:, target, direction, mode})` ×2 + (the absolute path is a tier-1 path match — no name collision). +5. Teardown the temp home + copy. + +### Step 0 — fixture AIS validation (gated on the live traversal; circularity) + +Before scoring, the harness reconciles each fixture against the live analyzer +(`metrics.mjs` is annotation-only; Step 0 is the *traversal* reconciliation): + +- the criterion must produce **≥ 1 PDG edge** (an accidental no-body / cap- + truncated criterion has unmeasurable ground truth → excluded, logged); +- the criterion symbol must **not** share `(filePath, startLine)` with another + `Function`/`Method` (one count query) — same-line projection ambiguity (R4) + would reconcile AIS against the wrong symbol's edges → excluded, logged. + +Per the **annotation-circularity guard**, this reconciliation runs *second*: the +AIS was written from source semantics *first* (U6), and Step 0 only confirms the +fixture is measurable substrate — it never derives ground truth from the +traversal. (When the harness's per-case recall surfaced a `direction`-vs-`AIS` +contradiction in `inter-pipeline-stages` — its AIS named callees while the +criterion was tagged `upstream` — the *fixture annotation* was corrected to +`downstream`, the direction its own AIS implies; the metric was not re-fit to a +traversal.) + +## Measured results (analyzer 1.6.7, 12 measurable + 1 excluded) + +| Scope | Mode | P | R | F1 | \|CIS\|/\|AIS\| | FPIS | FNIS | n | +|---|---|---|---|---|---|---|---|---| +| intra | callgraph | n/a | 0.000 | n/a | 0.000 | 0 | 6 | 6 | +| intra | pdg | n/a | 0.000 | n/a | 0.000 | 0 | 6 | 6 | +| inter | callgraph | 1.000 | 1.000 | 1.000 | 1.000 | 0 | 0 | 3 | +| inter | pdg | n/a | 0.000 | n/a | 0.000 | 0 | 9 | 3 | +| mixed | callgraph | 1.000 | 0.556 | 0.711 | 0.556 | 0 | 3 | 3 | +| mixed | pdg | n/a | 0.000 | n/a | 0.000 | 0 | 7 | 3 | + +Read it honestly: **call-graph mode is exact on the cross-function questions** +(inter P/R/F1 = 1.0; mixed precision 1.0, recall 0.556 because it cannot express +the intra criterion-self component). **PDG mode reports an empty symbol-level CIS +on every measurable fixture.** That is a real property of the shipped v1 +traversal, not a harness bug: PDG edges (CDG / REACHING_DEF) are *intra*- +procedural, connecting a function's own `BasicBlock`s; the traversal seeds on +**all** the criterion function's blocks and excludes seeds from the reachable +set, and the block→symbol projection collapses any intra reach back onto the +criterion itself. So at symbol granularity the intra-procedural blast radius is +∅ — PDG's v1 value is the **block-level detail** (`reachableBlocks` / `blockCount` +/ the per-edge-type reach), which the per-case lines surface (`pdg|blocks=…`), +not a symbol-level impact set. + +## Decision recommendation (the verdict — F2) + +> **Use `mode:'callgraph'` as the default** — it carries the inter-procedural +> reach the impact tool's safety question depends on (inter recall 1.0, mixed +> precision 1.0 on this corpus). **`mode:'pdg'` is an opt-in lens for +> intra-procedural dependence *inspection*** (its `reachableBlocks` / CDG + +> REACHING_DEF detail), valuable where the persisted PDG layer exists (`analyze +> --pdg`). It is **not** a replacement for, nor a strict improvement over, the +> call-graph blast radius: the two engines sit at different points on the +> precision/recall curve and neither strictly dominates. Promoting `mode:'pdg'` +> beyond opt-in is **gated on a `Function→BasicBlock` `CONTAINS_BLOCK` substrate +> edge** (deferred) that would let the symbol BFS chain natively into the PDG and +> give intra reach a symbol-level meaning — until then the symbol-level +> comparison is structurally one-sided and the harness says so rather than +> printing a flattering number. + +## Validity threats (the two that dominate — KTD9) + +1. **Ground-truth incompleteness.** A hand-annotated handful of fixtures yields + *point estimates* over a tiny, self-admittedly incomplete corpus. One + mis-annotation can swing F1 by a large fraction, so the harness reports + findings as a **direction**, not a headline decimal, and prints an explicit + "underpowered — directional only" banner when the corpus falls below the + floor. +2. **Annotation circularity.** PDG's `intra_AIS` risks being reconciled against + the PDG traversal's own output. **Mitigation:** these annotations are written + from SOURCE SEMANTICS first (U6) — reading the source and reasoning about + def→use / control dependence by hand — and reconciling against the live + traversal is the harness's **Step 0**, run *second*, only to confirm + measurability. Call-graph gets no such home-field annotation, so the + comparison is not rigged toward PDG. + +## Underpowered-corpus rule (F3) + +**Minimum corpus floor: ≥ 3 measurable cases per locus stratum, ≥ 12 total.** +Current corpus is exactly at the floor (intra 6, inter 3, mixed 3 = 12 +measurable; +1 excluded no-body). When the measurable count after exclusions +drops below the floor, the harness prints **"underpowered — directional only"** +and reports the DIRECTION ("PDG higher-precision on intra-scope") rather than +headline decimals — decimal precision (`F1 0.74 vs 0.68`) implies a confidence +the corpus cannot support. + +## Annotation fingerprint + `--check` (two gates, KTD10) + +`--check` runs **two non-byte-identity gates** (an exact-equality gate would go +perpetually red on legitimate accuracy changes): + +1. **One-sided F1 regression band** per mode per scope: fail iff `F1 < band − ε`; + improvements pass freely. `ε` and the per-`(scope,mode)` bands are versioned + in `baselines.json`. A `null` band means F1 is genuinely undefined for that + cell on this corpus (e.g. PDG's empty CIS) — the gate skips it. +2. **Order-independent annotation fingerprint** over the curated ground-truth + set (a SHA-256 over a sorted, line-collapsed canonicalization — mirrors the + `bench/cfg/measure.mjs` *technique*, written here, not a literal import). Any + unreviewed edit to a `ground-truth.json` (criterion, AIS membership, locus, + direction, edge kinds) trips it; a pure reordering of AIS entries does not. + +**Substrate stability (F5).** Real analyze is the repo's flaky lane, so `--check` +applies **median-of-K** across `GN_IMPACT_PDG_K` runs *before* comparing F1 to +the band, so substrate noise can't trip the metric gate. Default K = 1 (the +fixtures are tiny and deterministic in practice); raise it +(`GN_IMPACT_PDG_K=3`) in a flaky CI lane. + +## Runtime budget + +Each fixture costs **one full `analyze --pdg` child process** (a fresh tree-sitter +parse + CFG/PDG build + persist) plus two in-process `impact` calls. On these +tiny fixtures that is ≈ **3–6 s/fixture**, so the full 13-fixture corpus runs in +roughly **45–80 s** wall-clock single-threaded (K = 1). A K-fold `--check` +multiplies by K. For a fast substrate smoke, scope to a subset: +`--only=intra-dataflow-chain,inter-dispatcher-thin,mixed-guarded-dispatch` (or +`GN_IMPACT_PDG_ONLY=…`). Not wired into `npm test` (matches the other benches); +the deterministic metric-math unit test *is* in `npm test`. ## How to run -`measure.mjs` is **not yet built** (U7). When it lands: - -``` -node --import tsx gitnexus/bench/impact-pdg/measure.mjs # print the stratified report -node --import tsx gitnexus/bench/impact-pdg/measure.mjs --check # gate against baselines.json +```sh +cd gitnexus +node scripts/build.js # REQUIRED: workers spawn from dist/ +node --import tsx bench/impact-pdg/measure.mjs # print the stratified report + verdict +node --import tsx bench/impact-pdg/measure.mjs --json # machine report (for re-baselining) +node --import tsx bench/impact-pdg/measure.mjs --check # gate against baselines.json (exit non-zero on regression) +node --import tsx bench/impact-pdg/measure.mjs --only=a,b,c # fast subset (substrate smoke) ``` -The fixtures are validated today by the integration test +### Re-baseline (after a reviewed accuracy or ground-truth change) + +1. `node --import tsx bench/impact-pdg/measure.mjs --json` and read + `annotationFingerprint` + `strata[scope][mode].f1`. +2. Copy those into `baselines.json` (`annotationFingerprint`, the `f1Bands` + cells), bump `analyzerVersion` if the analyzer moved, adjust `epsilon` only + deliberately. +3. Confirm `--check` is green. + +The fixtures are also validated by the integration test `test/integration/impact-pdg-fixtures.test.ts` (schema well-formedness + a smoke -test that each fixture analyzes under `--pdg` and the criterion function -produces CDG + REACHING_DEF edges — a zero-edge criterion has unmeasurable +test that each fixture analyzes under `--pdg` and the criterion function produces +its declared CDG / REACHING_DEF edges — a zero-edge criterion has unmeasurable ground truth). diff --git a/gitnexus/bench/impact-pdg/baselines.json b/gitnexus/bench/impact-pdg/baselines.json new file mode 100644 index 000000000..72be784b5 --- /dev/null +++ b/gitnexus/bench/impact-pdg/baselines.json @@ -0,0 +1,21 @@ +{ + "_doc": "U7 impact-PDG accuracy baselines. Two NON-byte-identity gates (KTD10): (1) annotationFingerprint — an order-independent digest over the curated ground-truth set; any unreviewed edit to a ground-truth.json trips it (re-baseline deliberately after review). (2) f1Bands — a ONE-SIDED F1 regression band per mode per scope: a DROP below (band - epsilon) fails; improvements pass freely. A `null` band means F1 is genuinely undefined for that (scope,mode) on this corpus (e.g. PDG reports an empty intra/inter CIS, or a scope has no defined-F1 case) — the gate skips it (nothing to regress against). measure.mjs --check applies median-of-K across GN_IMPACT_PDG_K runs before comparing, so substrate flakiness (the real-analyze lane, F5) cannot trip the band. Re-baseline: `node --import tsx bench/impact-pdg/measure.mjs --json` → copy annotationFingerprint and strata[scope][mode].f1 here, bump analyzerVersion if the analyzer moved.", + "analyzerVersion": "1.6.7", + "epsilon": 0.05, + "annotationFingerprint": "853565db5f732b11fb452a90f92c065ab93913a32e7e83f886fd585f420ad516", + "f1Bands": { + "intra": { + "callgraph": null, + "pdg": null + }, + "inter": { + "callgraph": 1.0, + "pdg": null + }, + "mixed": { + "callgraph": 0.711, + "pdg": null + } + }, + "_f1BandsNote": "intra/* and *_pdg are null because the measured F1 is n/a on this corpus: PDG mode reports an empty symbol-level CIS for every measurable fixture (its block→symbol projection collapses a function's own dependence blocks onto the criterion, which the seed-exclusion drops), and callgraph intra-scope CIS is also empty (downstream/upstream over a self-contained function reaches no OTHER symbol). The only non-trivial F1 signal is callgraph on inter (1.0 — full cross-function recall) and mixed (0.711 — finds the callees, misses the intra criterion-self component). These two bands are the live regression guard; if a future change gives PDG a non-empty intra CIS (e.g. the deferred CONTAINS_BLOCK substrate edge), record its F1 here and the one-sided band starts guarding it." +} diff --git a/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/ground-truth.json b/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/ground-truth.json index 67733f399..dab28430a 100644 --- a/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/ground-truth.json +++ b/gitnexus/bench/impact-pdg/fixtures/inter-pipeline-stages/ground-truth.json @@ -3,7 +3,7 @@ "criterion": { "name": "runPipeline", "filePath": "src/pipeline.ts", - "direction": "upstream", + "direction": "downstream", "marker": "stageTransform(acc)", "pdgEdgeKinds": [ "REACHING_DEF", @@ -31,5 +31,5 @@ "note": "terminal stage" } ], - "rationale": "Direction is UPSTREAM: asking what `runPipeline` depends on. As a thin driver it threads `acc` through three stage calls (stageParse->stageTransform->stageEmit); the dependencies that matter cross function boundaries — the three stage functions — so inter_AIS holds them. intra_AIS is empty: the `acc` reassignments only carry delegate results, no independent computation. PDG mode is intra-procedural, so its inter-AIS recall here is ~0 BY DESIGN; the call-graph mode (upstream) reaches the stages. The `enabled` guard (added so the driver carries a CDG edge) keeps the criterion measurable for the smoke test without changing the cross-function locus. Distinct from the dispatcher (branch-routed) and facade (guarded-sequence) shapes — this is a straight pipeline driver." + "rationale": "Direction is DOWNSTREAM (dependencies): changing `runPipeline` affects the callees it invokes. In GitNexus `impact` vocabulary, downstream = dependencies/callees and upstream = dependants/callers; the stages are what the driver CALLS, so the correct tag is downstream (an earlier draft mislabeled this `upstream`, conflating the English 'upstream sources' with GitNexus's caller-direction — corrected in U7 after the harness surfaced a direction-vs-AIS contradiction). As a thin driver it threads `acc` through three stage calls (stageParse->stageTransform->stageEmit); the dependencies that matter cross function boundaries — the three stage functions — so inter_AIS holds them. intra_AIS is empty: the `acc` reassignments only carry delegate results, no independent computation. PDG mode is intra-procedural, so its inter-AIS recall here is ~0 BY DESIGN; the call-graph mode (downstream) reaches the stages. The `enabled` guard (added so the driver carries a CDG edge) keeps the criterion measurable for the smoke test without changing the cross-function locus. Distinct from the dispatcher (branch-routed) and facade (guarded-sequence) shapes — this is a straight pipeline driver." } diff --git a/gitnexus/bench/impact-pdg/measure.mjs b/gitnexus/bench/impact-pdg/measure.mjs new file mode 100644 index 000000000..ff1964127 --- /dev/null +++ b/gitnexus/bench/impact-pdg/measure.mjs @@ -0,0 +1,639 @@ +/** + * U7 — PDG-vs-call-graph impact ACCURACY measurement harness. + * + * Runs BOTH `impact` engines (`mode:'callgraph'` and `mode:'pdg'`) over the + * curated U6 ground-truth fixtures, computes precision/recall/F1 stratified by + * impact locus (intra / inter / mixed) plus cross-mode Jaccard + set-diffs, + * prints a stratified report ending in a plain-language DECISION RECOMMENDATION, + * and (under `--check`) gates regressions with two NON-byte-identity gates. + * + * ── Substrate (the load-bearing mechanism — KTD9/R8; plan U7 "Substrate + * decision") ────────────────────────────────────────────────────────────── + * `runPipelineFromRepo` is in-memory and never persists; `impact` queries a + * PERSISTED `repo.lbugPath` + a `meta.pdg` stamp. There is no exported + * `runAnalyze` (the entrypoint `analyzeCommand` calls `process.exit`, unusable + * in a loop), and the test-suite `vi.mock` registry bridge is vitest-only. So: + * REAL analyze via a temp `GITNEXUS_HOME`, mock-free. Per fixture: + * 1. point `process.env.GITNEXUS_HOME` at a per-run temp dir (honored by + * `repo-manager.getGlobalDir()` — it roots the registry; the per-repo DB + * lands in `/.gitnexus/`, so fixtures are copied to a temp + * working dir to keep the source tree clean); + * 2. SHELL OUT to the real CLI as a child process: + * node --import tsx src/cli/index.ts analyze --pdg --skip-git --index-only + * (child-process isolation sidesteps `process.exit`; real `saveMeta` + + * `registerRepo` land in the temp home; workers spawn from `dist/`, so the + * harness builds `dist/` first — run `node scripts/build.js`); + * 3. `new LocalBackend(); await init()` resolves the fixture via the REAL + * registry (the parent process ALSO sets `GITNEXUS_HOME` so init reads the + * temp registry, not the user's ~/.gitnexus); + * 4. `callTool('impact', {repo:, target, direction, mode})` ×2; + * 5. teardown the temp home + copy. + * + * The `repo` arg is the absolute fixture-copy PATH (tier-1 path match in + * `resolveRepoFromCache`) — unambiguous, no name collisions. + * + * ── Granularity / CIS-AIS framing ────────────────────────────────────────── + * See `metrics.mjs`. Symbol granularity, line-collapsed, order-independent. + * + * Build-free: `node --import tsx bench/impact-pdg/measure.mjs`. Runtime budget + * and re-baseline instructions: see README.md. + */ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { spawnSync } from 'node:child_process'; +import { fileURLToPath } from 'node:url'; + +import { + symbolKey, + toKeySet, + score, + compareModes, + aggregate, + partitionCisByScope, + aisByScope, + fingerprintAnnotationSet, + median, +} from './metrics.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const REPO_ROOT = path.resolve(__dirname, '..', '..'); // gitnexus/ +const FIXTURES_DIR = path.join(__dirname, 'fixtures'); +const BASELINE_PATH = path.join(__dirname, 'baselines.json'); +const CLI_ENTRY = path.join(REPO_ROOT, 'src', 'cli', 'index.ts'); + +const SCOPES = ['intra', 'inter', 'mixed']; +const MODES = ['callgraph', 'pdg']; + +// ── F3 minimum-corpus floor (KTD9): below this the harness reports DIRECTION +// only, never a headline decimal verdict. Mirrors the U6 schema test's floor. +const FLOOR_PER_STRATUM = 3; +const FLOOR_TOTAL = 12; + +const sha256 = (s) => crypto.createHash('sha256').update(s).digest('hex'); + +// ── fixture loading ──────────────────────────────────────────────────────── + +function loadFixtures(filter) { + const names = fs + .readdirSync(FIXTURES_DIR, { withFileTypes: true }) + .filter((d) => d.isDirectory()) + .map((d) => d.name) + .filter((n) => !filter || filter.includes(n)) + .sort(); + return names.map((name) => { + const dir = path.join(FIXTURES_DIR, name); + const gt = JSON.parse(fs.readFileSync(path.join(dir, 'ground-truth.json'), 'utf8')); + return { name, dir, gt, excluded: gt.pdgScoring === 'exclude' }; + }); +} + +// ── substrate: analyze a fixture into a temp GITNEXUS_HOME, run both modes ─── + +/** + * Copy the fixture src into a temp working dir, analyze it with `--pdg` as a + * child process (real persistence into the temp GITNEXUS_HOME), then drive both + * impact modes through a fresh LocalBackend. Returns the raw impact results + + * the working-copy path (so the criterion file paths line up with the + * annotations, which are repo-relative `src/...`). `pdgOn` toggles `--pdg` so + * the degraded-index scenario (KTD7) can be exercised. + */ +async function analyzeAndImpact(fx, home, { pdgOn = true } = {}) { + const work = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-impact-pdg-work-')); + fs.cpSync(path.join(fx.dir, 'src'), path.join(work, 'src'), { recursive: true }); + + const env = { ...process.env, GITNEXUS_HOME: home }; + const args = ['--import', 'tsx', CLI_ENTRY, 'analyze', work, '--skip-git', '--index-only']; + if (pdgOn) args.push('--pdg'); + const an = spawnSync(process.execPath, args, { + env, + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'pipe'], + timeout: 180000, + }); + if (an.status !== 0) { + fs.rmSync(work, { recursive: true, force: true }); + throw new Error( + `analyze failed for ${fx.name} (exit ${an.status}): ${(an.stderr || an.stdout || '').slice(-600)}`, + ); + } + + // The parent process must see the temp GITNEXUS_HOME too — LocalBackend.init() + // reads the REAL registry under getGlobalDir() (no mock). A fresh backend per + // fixture avoids cross-fixture pool/registry caching. + process.env.GITNEXUS_HOME = home; + const { LocalBackend } = await import(path.join(REPO_ROOT, 'src', 'mcp', 'local', 'local-backend.ts')); + const backend = new LocalBackend(); + await backend.init(); + + const results = {}; + for (const mode of MODES) { + results[mode] = await backend.callTool('impact', { + repo: work, + target: fx.gt.criterion.name, + direction: fx.gt.criterion.direction, + mode, + }); + } + return { work, results }; +} + +/** Flatten an impact result's byDepth into canonical symbol keys (the CIS). */ +function cisFromResult(res) { + const items = Object.values(res?.byDepth ?? {}).flat(); + const keys = new Set(); + const meta = { unresolved: 0, ambiguous: 0, blockCount: res?.blockCount ?? null }; + for (const it of items) { + if (it?.unresolved) { + meta.unresolved += 1; + // surfaced under its file as an unresolved shadow entry — kept in the CIS + // so a recall loss is never hidden, keyed by its file (no symbol name). + keys.add(symbolKey('(unresolved)', it.filePath)); + continue; + } + if (it?.ambiguous) meta.ambiguous += 1; + keys.add(symbolKey(it.name, it.filePath)); + } + return { keys, meta }; +} + +// ── Step 0: fixture AIS validation (gated on the live traversal; KTD9 +// circularity guard) ─────────────────────────────────────────────────────── + +/** + * Before scoring, reconcile each fixture's annotation against the LIVE analyzer: + * (a) the criterion must produce ≥1 PDG edge (no accidental no-body / cap + * truncation — a zero-edge criterion has unmeasurable ground truth); + * (b) the criterion symbol must NOT share `(filePath, startLine)` with another + * Function/Method (same-line projection ambiguity, R4) — one count query; + * (c) the annotation paths must line up with the analyzer's repo-relative + * paths (so symbol keys match across CIS/AIS). + * A fixture failing (a)/(b) is EXCLUDED from scoring and LOGGED (no silent cap). + */ +async function validateFixture(fx, work, exec) { + const lbugPath = path.join(work, '.gitnexus', 'lbug'); + // (a) criterion produces ≥1 PDG edge. Locate the criterion's blocks via the + // marker (the same technique the U6 smoke test uses) and count CDG/RD edges + // sourced inside them. + const marker = fx.gt.criterion.marker; + const blocks = await exec( + lbugPath, + `MATCH (b:BasicBlock) RETURN b.id AS id, b.text AS text`, + {}, + ); + const idsByAnchor = new Map(); + let anchor; + for (const b of blocks) { + const id = String(b.id ?? b[0] ?? ''); + const anc = id.slice(0, id.lastIndexOf(':')); + (idsByAnchor.get(anc) ?? idsByAnchor.set(anc, new Set()).get(anc)).add(id); + const text = String(b.text ?? b[1] ?? ''); + if (marker && text.includes(marker)) anchor = anc; + } + let critEdges = 0; + if (anchor) { + const blockIds = [...(idsByAnchor.get(anchor) ?? [])]; + if (blockIds.length > 0) { + const rows = await exec( + lbugPath, + `MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock) + WHERE r.type IN ['CDG','REACHING_DEF'] AND a.id IN $ids + RETURN count(r) AS n`, + { ids: blockIds }, + ); + critEdges = Number(rows?.[0]?.n ?? rows?.[0]?.[0] ?? 0); + } + } + + // (b) same-(filePath,startLine) collision for the criterion symbol (R4). + const collisionRows = await exec( + lbugPath, + `MATCH (s:\`Function\`) + WHERE s.name = $name AND s.filePath = $fp + RETURN s.startLine AS sl + UNION ALL + MATCH (s:\`Method\`) + WHERE s.name = $name AND s.filePath = $fp + RETURN s.startLine AS sl`, + { name: fx.gt.criterion.name, fp: fx.gt.criterion.filePath }, + ); + let sameLineCollision = false; + const startLine = collisionRows?.[0]?.sl ?? collisionRows?.[0]?.[0]; + if (startLine !== undefined && startLine !== null) { + const peers = await exec( + lbugPath, + `MATCH (s:\`Function\`) + WHERE s.filePath = $fp AND s.startLine = $sl + RETURN s.name AS name + UNION ALL + MATCH (s:\`Method\`) + WHERE s.filePath = $fp AND s.startLine = $sl + RETURN s.name AS name`, + { fp: fx.gt.criterion.filePath, sl: startLine }, + ); + sameLineCollision = (peers?.length ?? 0) > 1; + } + + const problems = []; + if (!anchor) problems.push(`criterion blocks not locatable via marker ${JSON.stringify(marker)}`); + if (critEdges === 0) problems.push('criterion produces ZERO PDG edges (unmeasurable ground truth)'); + if (sameLineCollision) + problems.push('criterion shares (filePath,startLine) with another Function/Method (R4 ambiguity)'); + return { critEdges, sameLineCollision, problems, measurable: problems.length === 0 }; +} + +// ── per-fixture scoring ────────────────────────────────────────────────────── + +/** + * Score one fixture for one mode, per scope. CIS partitioned into intra (the + * criterion symbol itself) / inter (others) / mixed (union); AIS likewise. + */ +function scoreFixtureMode(gt, cisKeys) { + const ais = aisByScope(gt); + const cisPart = partitionCisByScope(cisKeys, ais.criterionKey); + return { + intra: score(cisPart.intra, ais.intra), + inter: score(cisPart.inter, ais.inter), + mixed: score(cisPart.mixed, ais.mixed), + }; +} + +// ── reporting helpers ──────────────────────────────────────────────────────── + +const fmt = (v) => (v === null || v === undefined ? 'n/a' : Number(v).toFixed(3)); +const pad = (s, n) => String(s).padEnd(n); +const lpad = (s, n) => String(s).padStart(n); + +function renderTable(strata) { + const head = + `${pad('Scope', 7)} ${pad('Mode', 10)} ${lpad('P', 7)} ${lpad('R', 7)} ${lpad('F1', 7)} ` + + `${lpad('|CIS|/|AIS|', 11)} ${lpad('FPIS', 6)} ${lpad('FNIS', 6)} ${lpad('n', 4)}`; + const lines = [head, '-'.repeat(head.length)]; + for (const scope of SCOPES) { + for (const mode of MODES) { + const a = strata[scope][mode]; + lines.push( + `${pad(scope, 7)} ${pad(mode, 10)} ${lpad(fmt(a.precision), 7)} ${lpad(fmt(a.recall), 7)} ` + + `${lpad(fmt(a.f1), 7)} ${lpad(fmt(a.cisAisRatio), 11)} ${lpad(a.fpis, 6)} ${lpad(a.fnis, 6)} ` + + `${lpad(a.nCases, 4)}`, + ); + } + } + return lines.join('\n'); +} + +/** + * Plain-language DECISION RECOMMENDATION (F2 — the deliverable that answers + * "which is more accurate" as a verdict, not just a table). Derived from the + * measured numbers: compares inter-scope recall (the cross-function questions + * users most bring to impact) and any measured intra-scope precision edge. + */ +function decisionRecommendation(strata, underpowered, exclusions) { + const cgInterR = strata.inter.callgraph.recall; + const pdgInterR = strata.inter.pdg.recall; + const cgIntraP = strata.intra.callgraph.precision; + const pdgIntraP = strata.intra.pdg.precision; + const pdgIntraReports = strata.intra.pdg.nPrecision > 0; // did PDG report ANY intra symbol? + + const lines = []; + lines.push('DECISION RECOMMENDATION'); + if (underpowered) { + lines.push( + `Corpus is UNDERPOWERED (below the ${FLOOR_PER_STRATUM}/stratum, ${FLOOR_TOTAL}-total floor` + + ` after exclusions) — reporting DIRECTION, not headline decimals.`, + ); + } + + // Inter-scope: the cross-function blast radius. + if (cgInterR !== null && pdgInterR !== null) { + lines.push( + `On INTER-scope (cross-function) impact, call-graph recall is ${fmt(cgInterR)} vs PDG ${fmt(pdgInterR)}: ` + + `PDG's intra-procedural design means it recovers ~0 cross-function impact BY DESIGN (a capability ` + + `fact, not a defect). Call-graph is the correct engine for the "what else calls/uses this?" question.`, + ); + } + + // Intra-scope: the case PDG was built to win. + if (!pdgIntraReports) { + lines.push( + `On INTRA-scope, PDG mode reported NO owning symbols across the measurable corpus: its block→symbol ` + + `projection collapses a function's own dependence blocks back onto the criterion itself, which the ` + + `traversal excludes as the seed — so at SYMBOL granularity the intra-procedural blast radius is the ` + + `empty set. PDG's intra value in v1 is therefore the BLOCK-LEVEL detail it surfaces ` + + `(reachableBlocks / blockCount), NOT a symbol-level impact set. The harness records the per-fixture ` + + `dependence-block counts so this is visible, not hidden as a flat zero.`, + ); + } else if (pdgIntraP !== null && cgIntraP !== null) { + const verb = pdgIntraP > cgIntraP ? 'higher' : pdgIntraP < cgIntraP ? 'lower' : 'equal'; + lines.push( + `On INTRA-scope, PDG precision is ${fmt(pdgIntraP)} vs call-graph ${fmt(cgIntraP)} (${verb}). ` + + `This is the measured direction on this corpus, reported as a fact, not asserted as a hypothesis.`, + ); + } + + lines.push( + `VERDICT: use mode:'callgraph' as the default — it carries the inter-procedural reach that the ` + + `impact tool's safety question depends on. mode:'pdg' adds value as an OPT-IN lens for ` + + `intra-procedural dependence INSPECTION (its reachableBlocks / CDG+REACHING_DEF detail), and ` + + `where the persisted PDG layer exists (analyze --pdg). It is NOT a replacement for, nor a ` + + `strict improvement over, the call-graph blast radius: the two engines occupy different points ` + + `on the precision/recall curve and neither strictly dominates. Promotion of mode:'pdg' beyond ` + + `opt-in is GATED on a Function→BasicBlock CONTAINS_BLOCK substrate edge (deferred) that would let ` + + `the symbol BFS chain natively into the PDG and give intra reach a symbol-level meaning.`, + ); + if (exclusions.length > 0) { + lines.push(`Excluded from scoring: ${exclusions.map((e) => `${e.name} (${e.reason})`).join('; ')}.`); + } + return lines.join('\n'); +} + +// ── main run ───────────────────────────────────────────────────────────────── + +async function run() { + const CHECK = process.argv.includes('--check'); + const JSON_OUT = process.argv.includes('--json'); + // Optional subset for a fast substrate proof: --only=a,b,c or GN_IMPACT_PDG_ONLY=a,b + const onlyArg = process.argv.find((a) => a.startsWith('--only=')); + const onlyEnv = process.env.GN_IMPACT_PDG_ONLY; + const filter = (onlyArg ? onlyArg.slice('--only='.length) : onlyEnv || '') + .split(',') + .map((s) => s.trim()) + .filter(Boolean); + + const fixtures = loadFixtures(filter.length ? filter : null); + if (fixtures.length === 0) throw new Error('no fixtures found'); + + // K repeats for substrate-stability (F5). --check runs K times and gates on + // the per-(mode,scope) MEDIAN F1, so a flaky analyze edge cannot trip the band. + const K = CHECK ? Number(process.env.GN_IMPACT_PDG_K || 1) : 1; + + const home = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-impact-pdg-home-')); + const { initLbug, executeParameterized, closeLbug } = await import( + path.join(REPO_ROOT, 'src', 'core', 'lbug', 'pool-adapter.ts') + ); + + // exec wrapper that ensures the pool is initialised for Step 0's raw queries. + const initialised = new Set(); + const exec = async (lbugPath, q, p) => { + if (!initialised.has(lbugPath)) { + await initLbug(lbugPath, lbugPath).catch(() => {}); + initialised.add(lbugPath); + } + return executeParameterized(lbugPath, q, p); + }; + + const exclusions = []; + const perRunStrata = []; // K runs × { scope: { mode: aggregate } } + let perCaseDetail = null; // last run's per-case detail for the report + let degradedCheck = null; + + try { + for (let runIdx = 0; runIdx < K; runIdx++) { + // perCaseScores[scope][mode] = array of per-fixture score objects + const perScopeMode = {}; + for (const s of SCOPES) { + perScopeMode[s] = {}; + for (const m of MODES) perScopeMode[s][m] = []; + } + const detail = []; + + for (const fx of fixtures) { + if (fx.excluded) { + if (runIdx === 0) exclusions.push({ name: fx.name, reason: 'no-body (pdgScoring:exclude / KTD6)' }); + continue; + } + const { work, results } = await analyzeAndImpact(fx, home, { pdgOn: true }); + try { + // Step 0 — reconcile annotation against the live traversal. + const v = await validateFixture(fx, work, exec); + if (!v.measurable) { + if (runIdx === 0) + exclusions.push({ name: fx.name, reason: v.problems.join(' + ') }); + continue; + } + + const cg = cisFromResult(results.callgraph); + const pdg = cisFromResult(results.pdg); + const cgScores = scoreFixtureMode(fx.gt, cg.keys); + const pdgScores = scoreFixtureMode(fx.gt, pdg.keys); + + const locusScope = fx.gt.locus; // the stratum this fixture belongs to + // A fixture is scored in its OWN locus stratum (intra/inter/mixed). + if (SCOPES.includes(locusScope)) { + perScopeMode[locusScope].callgraph.push(cgScores[locusScope]); + perScopeMode[locusScope].pdg.push(pdgScores[locusScope]); + } + + if (runIdx === 0) { + const ais = aisByScope(fx.gt); + const cmp = compareModes(cg.keys, pdg.keys, ais.mixed); + detail.push({ + name: fx.name, + locus: fx.gt.locus, + criterion: fx.gt.criterion.name, + direction: fx.gt.criterion.direction, + critEdges: v.critEdges, + cg: { + count: results.callgraph.impactedCount, + symbols: [...cg.keys].sort(), + scores: cgScores, + }, + pdg: { + count: results.pdg.impactedCount, + blockCount: pdg.meta.blockCount, + unresolved: pdg.meta.unresolved, + ambiguous: pdg.meta.ambiguous, + symbols: [...pdg.keys].sort(), + scores: pdgScores, + }, + jaccard: cmp.jaccard, + pdgOnly: cmp.pdgOnly, + callgraphOnly: cmp.callgraphOnly, + }); + } + } finally { + await closeLbug(path.join(work, '.gitnexus', 'lbug')).catch(() => {}); + initialised.delete(path.join(work, '.gitnexus', 'lbug')); + fs.rmSync(work, { recursive: true, force: true }); + } + } + + // Aggregate this run's strata. + const strata = {}; + for (const s of SCOPES) { + strata[s] = {}; + for (const m of MODES) strata[s][m] = aggregate(perScopeMode[s][m]); + } + perRunStrata.push(strata); + if (runIdx === 0) perCaseDetail = detail; + } + + // ── Degraded-index check (KTD7): on ONE intra fixture, analyze WITHOUT + // --pdg and assert PDG mode reports a degradation note (skipped, not 0/0). + const degTarget = fixtures.find((f) => !f.excluded && f.gt.locus === 'intra'); + if (degTarget) { + const { work, results } = await analyzeAndImpact(degTarget, home, { pdgOn: false }); + try { + const pdgRes = results.pdg; + degradedCheck = { + name: degTarget.name, + pdgLayer: pdgRes.pdgLayer ?? null, + note: (pdgRes.note ?? pdgRes.error ?? '').slice(0, 140), + skipped: pdgRes.pdgLayer !== undefined && pdgRes.pdgLayer !== 'ready', + }; + } finally { + fs.rmSync(work, { recursive: true, force: true }); + } + } + } finally { + fs.rmSync(home, { recursive: true, force: true }); + } + + // ── Collapse K runs into the report strata: per (mode,scope) take the MEDIAN + // F1 across runs (F5 substrate stability); other fields from run 0. + const strata0 = perRunStrata[0]; + const report = {}; + for (const s of SCOPES) { + report[s] = {}; + for (const m of MODES) { + const f1s = perRunStrata.map((r) => r[s][m].f1).filter((v) => v !== null && v !== undefined); + const pmeds = perRunStrata.map((r) => r[s][m].precision).filter((v) => v !== null && v !== undefined); + const rmeds = perRunStrata.map((r) => r[s][m].recall).filter((v) => v !== null && v !== undefined); + report[s][m] = { + ...strata0[s][m], + f1: f1s.length ? median(f1s) : null, + precision: pmeds.length ? median(pmeds) : null, + recall: rmeds.length ? median(rmeds) : null, + }; + } + } + + // Underpowered floor (F3): measured cases per stratum after exclusions. + const measurableTotal = SCOPES.reduce( + (a, s) => a + Math.max(report[s].callgraph.nCases, report[s].pdg.nCases), + 0, + ); + const underpowered = + measurableTotal < FLOOR_TOTAL || + SCOPES.some((s) => Math.max(report[s].callgraph.nCases, report[s].pdg.nCases) < FLOOR_PER_STRATUM); + + const annotationFingerprint = fingerprintAnnotationSet(fixtures, sha256); + + const machineReport = { + analyzerVersion: JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf8')).version, + corpus: { total: fixtures.length, measurable: measurableTotal, excluded: exclusions }, + underpowered, + floor: { perStratum: FLOOR_PER_STRATUM, total: FLOOR_TOTAL }, + strata: report, + perCase: perCaseDetail, + degradedCheck, + annotationFingerprint, + runsK: K, + }; + + // ── output ────────────────────────────────────────────────────────────── + if (JSON_OUT) { + process.stdout.write(JSON.stringify(machineReport, null, 2) + '\n'); + } else { + const out = []; + out.push('=== impact-PDG accuracy report ==='); + out.push( + `analyzer ${machineReport.analyzerVersion} | corpus ${fixtures.length} ` + + `(${measurableTotal} measurable, ${exclusions.length} excluded) | runs K=${K}`, + ); + out.push(''); + out.push('Stratified P/R/F1 (symbol granularity, per impact locus):'); + out.push(renderTable(report)); + out.push(''); + out.push('Per-case Jaccard + cross-mode set-diffs (true = ∩AIS, noise = −AIS):'); + for (const d of perCaseDetail) { + out.push( + ` ${pad(d.name, 28)} locus=${pad(d.locus, 6)} J=${fmt(d.jaccard)} ` + + `cg|count=${d.cg.count} pdg|count=${d.pdg.count} pdg|blocks=${d.pdg.blockCount}`, + ); + if (d.callgraphOnly.all.length) + out.push( + ` callgraph-only: ${d.callgraphOnly.all.length} ` + + `(true ${d.callgraphOnly.true.length}, noise ${d.callgraphOnly.noise.length})`, + ); + if (d.pdgOnly.all.length) + out.push( + ` pdg-only: ${d.pdgOnly.all.length} ` + + `(true ${d.pdgOnly.true.length}, noise ${d.pdgOnly.noise.length})`, + ); + } + out.push(''); + if (degradedCheck) { + out.push( + `Degraded-index probe (KTD7): ${degradedCheck.name} analyzed WITHOUT --pdg → ` + + `pdgLayer=${degradedCheck.pdgLayer} skipped=${degradedCheck.skipped}`, + ); + out.push(` note: ${degradedCheck.note}`); + out.push(''); + } + out.push(`Annotation fingerprint: ${annotationFingerprint}`); + out.push(''); + out.push(decisionRecommendation(report, underpowered, exclusions)); + process.stdout.write(out.join('\n') + '\n'); + } + + // ── --check: two gates (KTD10) + F5 substrate stability ─────────────────── + if (CHECK) { + if (!fs.existsSync(BASELINE_PATH)) { + process.stderr.write(`[impact-pdg --check] FAIL: no baselines.json at ${BASELINE_PATH}\n`); + process.exit(1); + } + const baselines = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8')); + const failures = []; + + // Gate 1 — order-independent annotation fingerprint (unreviewed GT edits). + if (baselines.annotationFingerprint !== annotationFingerprint) { + failures.push( + `annotation fingerprint drift: ground-truth set changed without re-baseline ` + + `(got ${annotationFingerprint}, expected ${baselines.annotationFingerprint}) — ` + + `review the ground-truth.json edits, then re-baseline.`, + ); + } + + // Gate 2 — one-sided F1 regression band per mode per scope (improvements + // pass freely; only a DROP beyond ε fails). Median-of-K already applied. + const eps = baselines.epsilon ?? 0.05; + const bands = baselines.f1Bands ?? {}; + for (const s of SCOPES) { + for (const m of MODES) { + const baseF1 = bands[s]?.[m]; + const gotF1 = report[s][m].f1; + if (baseF1 === undefined || baseF1 === null) continue; // no band ⇒ nothing to regress against + if (gotF1 === null) { + // F1 became undefined where a baseline existed — a structural change + // (the scope lost all measurable cases). Flag it, don't pass silently. + failures.push( + `${s}/${m}: F1 is now n/a but baseline was ${fmt(baseF1)} (scope lost measurable cases?)`, + ); + continue; + } + if (gotF1 < baseF1 - eps) { + failures.push( + `${s}/${m}: F1 ${fmt(gotF1)} < baseline ${fmt(baseF1)} − ε(${eps}) = ${fmt(baseF1 - eps)} ` + + `(median of K=${K})`, + ); + } + } + } + + if (failures.length > 0) { + for (const f of failures) process.stderr.write(`[impact-pdg --check] FAIL: ${f}\n`); + process.exit(1); + } + process.stderr.write( + `[impact-pdg --check] PASS (${SCOPES.length} scopes × ${MODES.length} modes, ` + + `fingerprint OK, K=${K})\n`, + ); + } +} + +run().catch((err) => { + process.stderr.write(`[impact-pdg] ERROR: ${err?.stack || err}\n`); + process.exit(1); +}); diff --git a/gitnexus/bench/impact-pdg/metrics.mjs b/gitnexus/bench/impact-pdg/metrics.mjs new file mode 100644 index 000000000..7b2952355 --- /dev/null +++ b/gitnexus/bench/impact-pdg/metrics.mjs @@ -0,0 +1,238 @@ +/** + * Pure scorer + annotation canonicalizer for the impact-PDG accuracy harness + * (U7). NO substrate here — no `runPipelineFromRepo`, no `LocalBackend`, no DB, + * no child-process `analyze`. Everything in this module is a pure function over + * plain symbol-set inputs, so the metric-math unit test + * (`test/unit/impact-pdg-metric-math.test.ts`) can import and assert the + * arithmetic deterministically, staying OUT of the flaky full-pipeline lane + * (Arch-review Issue 5). `measure.mjs` imports these for the live loop. + * + * ── CIS / AIS framing (KTD9 — Arnold–Bohner) ─────────────────────────────── + * CIS = Computed Impact Set: the symbols a mode REPORTS as impacted. + * AIS = Actual Impact Set: the curated ground-truth symbols truly affected. + * precision = |AIS∩CIS| / |CIS| (over-approximation cost; ∅ CIS ⇒ undefined) + * recall = |AIS∩CIS| / |AIS| (under-approximation; ∅ AIS ⇒ undefined) + * F1 = harmonic mean (undefined if either is undefined) + * FPIS = CIS − AIS (false positives — noise) + * FNIS = AIS − CIS (false negatives — the DANGEROUS miss for a safety tool) + * + * ── Granularity (locked) ─────────────────────────────────────────────────── + * Symbol granularity, NEVER block-id (block ids carry fragile fnLine:fnCol:idx). + * A symbol key is `@` — order-independent, line-collapsed. An + * `intra_AIS` entry (statement-granular: lines within the criterion function) + * collapses to its OWNING symbol; an `inter_AIS` entry already names a whole + * symbol. This is why per-scope intra-AIS is the singleton {criterion}. + */ + +/** Order-independent symbol key. Collapses statement lines onto their symbol. */ +export function symbolKey(symbol, filePath) { + return `${symbol}@${filePath}`; +} + +/** Canonicalize an iterable of {symbol,filePath} (or pre-made keys) → a Set. */ +export function toKeySet(entries) { + const out = new Set(); + for (const e of entries) { + if (typeof e === 'string') out.add(e); + else out.add(symbolKey(e.symbol, e.filePath)); + } + return out; +} + +function intersectionSize(a, b) { + let n = 0; + const [small, large] = a.size <= b.size ? [a, b] : [b, a]; + for (const x of small) if (large.has(x)) n++; + return n; +} + +/** a − b as a sorted array of keys. */ +export function difference(a, b) { + const out = []; + for (const x of a) if (!b.has(x)) out.push(x); + return out.sort(); +} + +/** + * Core CIS-vs-AIS scorer. `cis` / `ais` are Sets of canonical symbol keys. + * + * Empty-denominator semantics are EXPLICIT (not silently 0 or 1): + * - |CIS|=0 ⇒ precision = null (no predictions to be right/wrong about). + * - |AIS|=0 ⇒ recall = null (nothing to find — this scope has no truth). + * - F1 = null whenever precision or recall is null OR both are 0. + * A null metric is REPORTED as `n/a`, never averaged in as 0 — collapsing it to + * 0 would punish a mode for a scope that simply has no ground truth (the + * apples-to-oranges trap, R1). + */ +export function score(cis, ais) { + const tp = intersectionSize(cis, ais); + const precision = cis.size === 0 ? null : tp / cis.size; + const recall = ais.size === 0 ? null : tp / ais.size; + let f1 = null; + if (precision !== null && recall !== null && precision + recall > 0) { + f1 = (2 * precision * recall) / (precision + recall); + } + return { + tp, + cisSize: cis.size, + aisSize: ais.size, + precision, + recall, + f1, + fpis: difference(cis, ais), // CIS − AIS (noise / over-approx) + fnis: difference(ais, cis), // AIS − CIS (missed / under-approx) + fpisCount: cis.size - tp, + fnisCount: ais.size - tp, + // |CIS|/|AIS| size ratio (>1 over-approximates, <1 under). null if |AIS|=0. + cisAisRatio: ais.size === 0 ? null : cis.size / ais.size, + }; +} + +/** + * Cross-mode comparison of two CIS sets against a shared AIS (KTD9 set-diffs). + * Jaccard(callgraph_CIS, pdg_CIS) + directional set-diffs, each split into + * `true` (∩AIS — a real find the other mode missed) vs `noise` (−AIS — a false + * positive the other mode avoided). + */ +export function compareModes(callgraphCis, pdgCis, ais) { + const union = new Set([...callgraphCis, ...pdgCis]); + const inter = intersectionSize(callgraphCis, pdgCis); + const jaccard = union.size === 0 ? null : inter / union.size; + + const pdgOnly = difference(pdgCis, callgraphCis); + const callgraphOnly = difference(callgraphCis, pdgCis); + const splitByAis = (keys) => { + const trueFinds = keys.filter((k) => ais.has(k)).sort(); + const noise = keys.filter((k) => !ais.has(k)).sort(); + return { all: keys, true: trueFinds, noise }; + }; + return { + jaccard, + intersectionSize: inter, + unionSize: union.size, + pdgOnly: splitByAis(pdgOnly), + callgraphOnly: splitByAis(callgraphOnly), + }; +} + +/** + * Aggregate per-case scores for ONE (mode, scope) into a corpus row. Averaging + * follows KTD9 "per change, averaged over the corpus": a case with a null metric + * (e.g. |CIS|=0 precision) is EXCLUDED from that metric's mean (counted in + * `nMetric`), never folded in as 0. The macro-average is over the cases that + * actually have the metric defined; `nCases` records the stratum size for the + * underpowered-corpus floor (F3). + */ +export function aggregate(perCaseScores) { + const avg = (sel) => { + const xs = perCaseScores.map(sel).filter((v) => v !== null && v !== undefined); + if (xs.length === 0) return { mean: null, n: 0 }; + return { mean: xs.reduce((a, b) => a + b, 0) / xs.length, n: xs.length }; + }; + const p = avg((s) => s.precision); + const r = avg((s) => s.recall); + const f = avg((s) => s.f1); + const ratio = avg((s) => s.cisAisRatio); + return { + nCases: perCaseScores.length, + precision: p.mean, + nPrecision: p.n, + recall: r.mean, + nRecall: r.n, + f1: f.mean, + nF1: f.n, + cisAisRatio: ratio.mean, + // Summed FPIS/FNIS counts over the stratum (totals, not means) — the + // absolute over/under-approximation volume. + fpis: perCaseScores.reduce((a, s) => a + (s.fpisCount ?? 0), 0), + fnis: perCaseScores.reduce((a, s) => a + (s.fnisCount ?? 0), 0), + }; +} + +/** + * Partition a mode's reported CIS keys into per-scope sub-CIS, given the + * criterion's own symbol key. INTRA = the criterion symbol itself (the only + * symbol whose blocks/edges are intra-procedural); INTER = every OTHER reported + * symbol (callees / cross-function reach). `unresolved` shadow entries (id null, + * surfaced under a file) are kept in INTER — they are non-criterion reach the + * mode could not attribute to a named symbol, and dropping them would hide a + * recall fact (R9). MIXED scope unions both. + */ +export function partitionCisByScope(cisKeys, criterionKey) { + const intra = new Set(); + const inter = new Set(); + for (const k of cisKeys) { + if (k === criterionKey) intra.add(k); + else inter.add(k); + } + return { intra, inter, mixed: new Set([...intra, ...inter]) }; +} + +/** + * Build the scope-appropriate AIS key sets from a ground-truth record. + * - intra: the criterion symbol itself (intra_AIS lines collapse onto it). A + * case with a non-empty intra_AIS contributes {criterion}; an empty intra_AIS + * contributes ∅ (no intra truth → recall n/a, not 0). + * - inter: the distinct callee symbols named in inter_AIS. + * - mixed: the union. + * Keys are `@` with paths normalised to the criterion's path + * style (the fixture annotations and the analyzer both use repo-relative + * `src/...` paths, so no rewrite is needed — asserted by Step 0). + */ +export function aisByScope(gt) { + const critKey = symbolKey(gt.criterion.name, gt.criterion.filePath); + const intra = new Set(); + if (Array.isArray(gt.intra_AIS) && gt.intra_AIS.length > 0) intra.add(critKey); + const inter = toKeySet((gt.inter_AIS ?? []).map((e) => ({ symbol: e.symbol, filePath: e.filePath }))); + return { criterionKey: critKey, intra, inter, mixed: new Set([...intra, ...inter]) }; +} + +/** + * Order-independent annotation-set fingerprint (KTD10). Mirrors the + * bench/cfg/measure.mjs canonicalization TECHNIQUE (sort every collection, + * stringify deterministically, hash) — but is annotation-set-shaped and written + * here, NOT a literal import of `canonicalizeCfg`. Any unreviewed edit to a + * ground-truth.json (criterion, AIS membership, locus, direction, edge kinds) + * changes the digest, tripping a `--check` gate distinct from the F1 band. + * + * `hash` is injected (node:crypto in the harness; a stub in the unit test) so + * this module pulls no node-only deps that would complicate the test import. + */ +export function canonicalizeAnnotationSet(fixtures) { + const canonAis = (entries) => + (entries ?? []) + .map((e) => `${e.symbol}|${e.filePath}|${e.line ?? '-'}`) + .sort() + .join(';'); + const lines = fixtures + .map((fx) => { + const c = fx.gt.criterion; + const kinds = Array.isArray(c.pdgEdgeKinds) ? [...c.pdgEdgeKinds].sort().join(',') : '-'; + return [ + `case=${fx.name}`, + `schema=${fx.gt.schemaVersion}`, + `crit=${c.name}|${c.filePath}|${c.direction}|${c.marker ?? '-'}|${kinds}`, + `locus=${fx.gt.locus}`, + `pdgScoring=${fx.gt.pdgScoring ?? '-'}`, + `provenance=${fx.gt.provenance}`, + `intra=${canonAis(fx.gt.intra_AIS)}`, + `inter=${canonAis(fx.gt.inter_AIS)}`, + ].join('\n'); + }) + .sort() + .join('\n====\n'); + return lines; +} + +/** SHA-256 the canonical string with an injected hashing function. */ +export function fingerprintAnnotationSet(fixtures, sha256Hex) { + return sha256Hex(canonicalizeAnnotationSet(fixtures)); +} + +/** median of a numeric array (substrate-stability gate, F5). */ +export function median(xs) { + if (xs.length === 0) return null; + const s = [...xs].sort((a, b) => a - b); + const m = Math.floor(s.length / 2); + return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2; +} diff --git a/gitnexus/test/unit/impact-pdg-metric-math.test.ts b/gitnexus/test/unit/impact-pdg-metric-math.test.ts new file mode 100644 index 000000000..365edea30 --- /dev/null +++ b/gitnexus/test/unit/impact-pdg-metric-math.test.ts @@ -0,0 +1,263 @@ +// U7 — metric-math unit test for the impact-PDG accuracy scorer. +// +// Asserts the scorer arithmetic (precision / recall / F1 / Jaccard / set-diffs / +// aggregation / annotation fingerprint) on SYNTHETIC CIS/AIS sets ONLY — no +// `runPipelineFromRepo`, no `analyze`, no `LocalBackend`, no DB. The pure +// scorer lives in `bench/impact-pdg/metrics.mjs`, imported here directly, so +// this test is deterministic and stays OUT of the flaky full-pipeline lane +// (Arch-review Issue 5). The live substrate is exercised manually by +// `measure.mjs`, never in `npm test`. + +import { describe, it, expect } from 'vitest'; +// @ts-expect-error — .mjs pure-JS module, no types; intentional (build-free harness). +import * as M from '../../bench/impact-pdg/metrics.mjs'; + +const k = (sym: string, file = 'src/a.ts') => M.symbolKey(sym, file); +const setOf = (...syms: string[]) => M.toKeySet(syms.map((s) => k(s))); + +describe('impact-pdg metric math — score()', () => { + it('computes precision/recall/F1 on a known partial overlap', () => { + // CIS = {a,b,c}, AIS = {b,c,d}. TP = {b,c} = 2. + const cis = setOf('a', 'b', 'c'); + const ais = setOf('b', 'c', 'd'); + const s = M.score(cis, ais); + expect(s.tp).toBe(2); + expect(s.precision).toBeCloseTo(2 / 3, 12); // 2 of 3 predicted are real + expect(s.recall).toBeCloseTo(2 / 3, 12); // 2 of 3 real are found + expect(s.f1).toBeCloseTo(2 / 3, 12); // p==r ⇒ F1==p + expect(s.fpis).toEqual([k('a')]); // CIS−AIS + expect(s.fnis).toEqual([k('d')]); // AIS−CIS + expect(s.fpisCount).toBe(1); + expect(s.fnisCount).toBe(1); + expect(s.cisAisRatio).toBeCloseTo(1, 12); + }); + + it('perfect match ⇒ P=R=F1=1, empty diffs', () => { + const s = M.score(setOf('a', 'b'), setOf('a', 'b')); + expect(s.precision).toBe(1); + expect(s.recall).toBe(1); + expect(s.f1).toBe(1); + expect(s.fpis).toEqual([]); + expect(s.fnis).toEqual([]); + }); + + it('asymmetric F1: high recall, low precision', () => { + // CIS over-approximates: {a,b,c,d}, AIS = {a}. TP=1. + const s = M.score(setOf('a', 'b', 'c', 'd'), setOf('a')); + expect(s.precision).toBeCloseTo(1 / 4, 12); + expect(s.recall).toBe(1); + // F1 = 2*(0.25*1)/(0.25+1) = 0.5/1.25 = 0.4 + expect(s.f1).toBeCloseTo(0.4, 12); + expect(s.cisAisRatio).toBeCloseTo(4, 12); // 4× over-approx + expect(s.fpisCount).toBe(3); + expect(s.fnisCount).toBe(0); + }); + + it('disjoint sets ⇒ P=R=F1=0', () => { + const s = M.score(setOf('a', 'b'), setOf('c', 'd')); + expect(s.precision).toBe(0); + expect(s.recall).toBe(0); + expect(s.f1).toBe(null); // p+r==0 ⇒ harmonic mean undefined, reported n/a + expect(s.fnis).toEqual([k('c'), k('d')]); + }); + + it('empty CIS ⇒ precision n/a (null), recall 0, F1 n/a (the PDG-intra case)', () => { + // This is the SHAPE the real harness measures for PDG on a self-contained + // function: the mode reports nothing, AIS = {criterion}. precision is + // genuinely undefined (no predictions), recall is 0 (missed everything). + const s = M.score(new Set(), setOf('criterion')); + expect(s.precision).toBe(null); // |CIS|=0 ⇒ undefined, NOT 0 + expect(s.recall).toBe(0); + expect(s.f1).toBe(null); + expect(s.fnis).toEqual([k('criterion')]); // the dangerous miss + expect(s.cisAisRatio).toBe(0); + }); + + it('empty AIS ⇒ recall n/a (null) — a scope with no ground truth', () => { + const s = M.score(setOf('a'), new Set()); + expect(s.recall).toBe(null); // |AIS|=0 ⇒ undefined, NOT 0 + expect(s.precision).toBe(0); // predicted a, none real + expect(s.f1).toBe(null); + expect(s.cisAisRatio).toBe(null); + }); +}); + +describe('impact-pdg metric math — compareModes()', () => { + it('Jaccard + directional set-diffs split true/noise', () => { + // callgraph finds {a,b,c} (a,b real, c noise); pdg finds {b,d} (b real, d noise). + // AIS = {a,b,e}. + const cg = setOf('a', 'b', 'c'); + const pdg = setOf('b', 'd'); + const ais = setOf('a', 'b', 'e'); + const cmp = M.compareModes(cg, pdg, ais); + // union {a,b,c,d}=4, inter {b}=1 ⇒ Jaccard 1/4. + expect(cmp.jaccard).toBeCloseTo(0.25, 12); + expect(cmp.intersectionSize).toBe(1); + expect(cmp.unionSize).toBe(4); + // pdg-only = {d}; d ∉ AIS ⇒ noise. + expect(cmp.pdgOnly.all).toEqual([k('d')]); + expect(cmp.pdgOnly.true).toEqual([]); + expect(cmp.pdgOnly.noise).toEqual([k('d')]); + // callgraph-only = {a,c}; a ∈ AIS (true find pdg missed), c ∉ AIS (noise). + expect(cmp.callgraphOnly.all).toEqual([k('a'), k('c')]); + expect(cmp.callgraphOnly.true).toEqual([k('a')]); + expect(cmp.callgraphOnly.noise).toEqual([k('c')]); + }); + + it('two empty CIS ⇒ Jaccard n/a (null), no diffs', () => { + const cmp = M.compareModes(new Set(), new Set(), setOf('a')); + expect(cmp.jaccard).toBe(null); + expect(cmp.pdgOnly.all).toEqual([]); + expect(cmp.callgraphOnly.all).toEqual([]); + }); +}); + +describe('impact-pdg metric math — partitionCisByScope() / aisByScope()', () => { + it('partitions a CIS into intra (=criterion) vs inter (others)', () => { + const critKey = k('route', 'src/mixed.ts'); + const cis = M.toKeySet([ + k('route', 'src/mixed.ts'), // the criterion itself ⇒ intra + k('fast', 'src/mixed.ts'), // a callee ⇒ inter + k('slow', 'src/mixed.ts'), // a callee ⇒ inter + ]); + const part = M.partitionCisByScope(cis, critKey); + expect([...part.intra]).toEqual([critKey]); + expect([...part.inter].sort()).toEqual([k('fast', 'src/mixed.ts'), k('slow', 'src/mixed.ts')]); + expect(part.mixed.size).toBe(3); + }); + + it('aisByScope collapses intra_AIS lines onto the criterion symbol', () => { + const gt = { + criterion: { name: 'route', filePath: 'src/mixed.ts', direction: 'downstream' }, + intra_AIS: [ + { symbol: 'route', filePath: 'src/mixed.ts', line: 16 }, + { symbol: 'route', filePath: 'src/mixed.ts', line: 18 }, + { symbol: 'route', filePath: 'src/mixed.ts', line: 20 }, + ], + inter_AIS: [ + { symbol: 'fast', filePath: 'src/mixed.ts' }, + { symbol: 'slow', filePath: 'src/mixed.ts' }, + ], + }; + const a = M.aisByScope(gt); + // three intra lines collapse to the singleton {criterion}. + expect([...a.intra]).toEqual([k('route', 'src/mixed.ts')]); + expect([...a.inter].sort()).toEqual([k('fast', 'src/mixed.ts'), k('slow', 'src/mixed.ts')]); + expect(a.mixed.size).toBe(3); + }); + + it('aisByScope: empty intra_AIS ⇒ empty intra scope (no false {criterion})', () => { + const gt = { + criterion: { name: 'dispatch', filePath: 'src/d.ts', direction: 'downstream' }, + intra_AIS: [], + inter_AIS: [{ symbol: 'handleA', filePath: 'src/d.ts' }], + }; + const a = M.aisByScope(gt); + expect(a.intra.size).toBe(0); // no intra truth ⇒ recall will be n/a, not 0 + expect([...a.inter]).toEqual([k('handleA', 'src/d.ts')]); + }); +}); + +describe('impact-pdg metric math — aggregate()', () => { + it('macro-averages defined metrics, EXCLUDING nulls (not folding as 0)', () => { + const per = [ + { precision: 1, recall: 1, f1: 1, cisAisRatio: 1, fpisCount: 0, fnisCount: 0 }, + { precision: 0.5, recall: 1, f1: 2 / 3, cisAisRatio: 2, fpisCount: 1, fnisCount: 0 }, + // a null-precision case (|CIS|=0): excluded from the precision mean. + { precision: null, recall: 0, f1: null, cisAisRatio: 0, fpisCount: 0, fnisCount: 2 }, + ]; + const agg = M.aggregate(per); + expect(agg.nCases).toBe(3); + // precision mean over the 2 defined cases = (1+0.5)/2 = 0.75 + expect(agg.precision).toBeCloseTo(0.75, 12); + expect(agg.nPrecision).toBe(2); + // recall mean over all 3 (none null) = (1+1+0)/3 + expect(agg.recall).toBeCloseTo(2 / 3, 12); + expect(agg.nRecall).toBe(3); + // F1 mean over the 2 defined = (1 + 2/3)/2 + expect(agg.f1).toBeCloseTo((1 + 2 / 3) / 2, 12); + expect(agg.nF1).toBe(2); + expect(agg.fpis).toBe(1); // summed totals + expect(agg.fnis).toBe(2); + }); + + it('all-null stratum ⇒ null means, n=0 (reported n/a)', () => { + const agg = M.aggregate([{ precision: null, recall: null, f1: null, cisAisRatio: null }]); + expect(agg.precision).toBe(null); + expect(agg.recall).toBe(null); + expect(agg.f1).toBe(null); + expect(agg.nF1).toBe(0); + }); +}); + +describe('impact-pdg metric math — annotation fingerprint (KTD10)', () => { + const fakeHash = (s: string): string => { + // tiny deterministic non-crypto digest — enough to assert drift sensitivity + // without pulling node:crypto into the unit (the real harness injects sha256). + let h = 5381; + for (let i = 0; i < s.length; i++) h = ((h << 5) + h + s.charCodeAt(i)) >>> 0; + return h.toString(16); + }; + const fx = (over: Record = {}) => ({ + name: 'c1', + gt: { + schemaVersion: 1, + criterion: { + name: 'f', + filePath: 'src/f.ts', + direction: 'downstream', + marker: 'x', + pdgEdgeKinds: ['REACHING_DEF'], + }, + locus: 'intra', + provenance: 'manual', + intra_AIS: [{ symbol: 'f', filePath: 'src/f.ts', line: 3 }], + inter_AIS: [], + ...over, + }, + }); + + it('is order-independent over the fixture list', () => { + const a = M.fingerprintAnnotationSet([fx({}), { ...fx({}), name: 'c2' }], fakeHash); + const b = M.fingerprintAnnotationSet([{ ...fx({}), name: 'c2' }, fx({})], fakeHash); + expect(a).toBe(b); + }); + + it('trips when an AIS membership changes (catches unreviewed ground-truth edits)', () => { + const base = M.fingerprintAnnotationSet([fx({})], fakeHash); + const edited = M.fingerprintAnnotationSet( + [fx({ intra_AIS: [{ symbol: 'f', filePath: 'src/f.ts', line: 99 }] })], + fakeHash, + ); + expect(edited).not.toBe(base); + }); + + it('trips when the criterion direction flips', () => { + const base = M.fingerprintAnnotationSet([fx({})], fakeHash); + const flipped = M.fingerprintAnnotationSet( + [fx({ criterion: { name: 'f', filePath: 'src/f.ts', direction: 'upstream', marker: 'x', pdgEdgeKinds: ['REACHING_DEF'] } })], + fakeHash, + ); + expect(flipped).not.toBe(base); + }); + + it('is STABLE under a pure reordering of AIS entries within a case', () => { + const a = M.fingerprintAnnotationSet( + [fx({ intra_AIS: [{ symbol: 'f', filePath: 'src/f.ts', line: 3 }, { symbol: 'f', filePath: 'src/f.ts', line: 5 }] })], + fakeHash, + ); + const b = M.fingerprintAnnotationSet( + [fx({ intra_AIS: [{ symbol: 'f', filePath: 'src/f.ts', line: 5 }, { symbol: 'f', filePath: 'src/f.ts', line: 3 }] })], + fakeHash, + ); + expect(a).toBe(b); + }); +}); + +describe('impact-pdg metric math — median (substrate-stability gate F5)', () => { + it('odd/even/empty', () => { + expect(M.median([3, 1, 2])).toBe(2); + expect(M.median([4, 1, 3, 2])).toBe(2.5); + expect(M.median([])).toBe(null); + }); +});