diff --git a/gitnexus/bench/impact-pdg/README.md b/gitnexus/bench/impact-pdg/README.md index d50d94785..ef66bbd47 100644 --- a/gitnexus/bench/impact-pdg/README.md +++ b/gitnexus/bench/impact-pdg/README.md @@ -7,9 +7,13 @@ > to run*). The harness drives both `impact` engines over the fixtures — PDG > **seeded on the criterion's statement line** so it returns the dependence slice > — prints a stratified P/R/F1 table + a plain-language decision recommendation, -> and gates regressions with `--check`. The measured result: **PDG is exact at -> intra-procedural statement granularity; call-graph is exact at inter-procedural -> symbol granularity; the two answer different questions and neither dominates.** +> and gates regressions with `--check`. It now also prints an additive **unified +> impact axes** table that keeps line-level and symbol-level truth separate while +> comparing current `callgraph`, current `pdg`, and the evaluation-only +> `composed-current` baseline. The measured native result remains: **PDG is exact +> at intra-procedural statement granularity; call-graph is exact at +> inter-procedural symbol granularity; the two answer different questions and +> neither dominates.** ## What this measures @@ -32,6 +36,33 @@ granularity against its native ground truth and reports both side by side. The single blended number — and the answer is *they answer different questions; neither strictly dominates*. +## Unified impact axes + +The harness also reports a separate unified comparison that is designed for the +next architecture question: *could a future PDG-only / SDG-like impact engine +replace the composition of today's engines?* This report is additive. It does not +replace the native table above, and it does not change `baselines.json` gating. + +Unified AIS has two namespaces: + +- `statement::` for intra-procedural line truth from `intra_AIS` +- `symbol:@` for inter-procedural symbol truth from `inter_AIS` + +Each engine is adapted onto those axes without lossy projection: + +- `callgraph` contributes only the `symbol` axis. +- `pdg` contributes only the `statement` axis. +- `composed-current` is an evaluation-only control row that unions current + callgraph symbols with current PDG statements. + +The report intentionally has no single blended unified F1. A future +`pdg-interproc` or SDG candidate must be judged axis-by-axis against +`composed-current` so line precision cannot hide inter-symbol misses, and +symbol recall cannot hide statement-level blindness. The control row is a recall +baseline, not a perfection claim: current PDG can still contribute intra-line +noise on pure-inter fixtures, so a future SDG candidate should match or exceed +recall while reducing or bounding FPIS. + > **A note on `line`.** A whole-symbol PDG slice (no `line`) is empty by design: > intra-procedural dependence stays inside the function, so every reachable block > is already part of the whole-symbol seed. The useful PDG mode is the @@ -299,9 +330,13 @@ Read it honestly: > **cannot answer at all** (it has no notion of a statement). > > They **compose**: a full mixed-locus blast radius is the *union* of -> call-graph's inter-symbol reach and PDG's intra-statement slice. Reach for the -> line-seeded PDG when you need statement-level dependence *inside* a function; -> reach for call-graph when you need *cross-function* reach. The earlier verdict +> call-graph's inter-symbol reach and PDG's intra-statement slice. The unified +> axes table makes that composition explicit through the `composed-current` row, +> which is the recall baseline a future SDG / `pdg-interproc` candidate must +> match or exceed while reducing or bounding FPIS. Reach for the line-seeded +> PDG when you need statement-level dependence *inside* a function; reach for +> call-graph when you need +> *cross-function* reach. The earlier verdict > ("PDG is empty / call-graph wins") was an artifact of the **whole-symbol** seed > — a whole-symbol slice has nothing to report because intra-procedural dependence > never leaves the function. Seeding the changed *statement* is what makes PDG's diff --git a/gitnexus/bench/impact-pdg/measure.mjs b/gitnexus/bench/impact-pdg/measure.mjs index 1e872223f..cfe2e8784 100644 --- a/gitnexus/bench/impact-pdg/measure.mjs +++ b/gitnexus/bench/impact-pdg/measure.mjs @@ -62,6 +62,12 @@ import { score, aggregate, aisByScope, + unifiedAis, + callgraphUnifiedCis, + pdgUnifiedCis, + composeUnifiedCis, + scoreUnifiedAxes, + aggregateUnifiedScores, fingerprintAnnotationSet, median, } from './metrics.mjs'; @@ -74,6 +80,7 @@ const CLI_ENTRY = path.join(REPO_ROOT, 'src', 'cli', 'index.ts'); const SCOPES = ['intra', 'inter', 'mixed']; const MODES = ['callgraph', 'pdg']; +const UNIFIED_MODES = ['callgraph', 'pdg', 'composed-current']; // ── F3 minimum-corpus floor (KTD9): below this the harness reports DIRECTION // only, never a headline decimal verdict. Mirrors the U6 schema test's floor. @@ -355,6 +362,35 @@ function renderTable(strata) { return lines.join('\n'); } +function renderUnifiedTable(unified) { + const head = + `${pad('Mode', 17)} ${pad('Axis', 13)} ${lpad('P', 7)} ${lpad('R', 7)} ${lpad('F1', 7)} ` + + `${lpad('|CIS|/|AIS|', 11)} ${lpad('FPIS', 6)} ${lpad('FNIS', 6)} ${lpad('n', 4)}`; + const lines = [head, '-'.repeat(head.length)]; + const axes = [ + ['intraLine', 'line/intra'], + ['interSymbol', 'symbol/inter'], + ]; + for (const mode of UNIFIED_MODES) { + for (const [axis, label] of axes) { + const a = unified[mode][axis]; + lines.push( + `${pad(mode, 17)} ${pad(label, 13)} ${lpad(fmt(a.precision), 7)} ${lpad(fmt(a.recall), 7)} ` + + `${lpad(fmt(a.f1), 7)} ${lpad(fmt(a.cisAisRatio), 11)} ${lpad(a.fpis, 6)} ${lpad(a.fnis, 6)} ` + + `${lpad(a.nCases, 4)}`, + ); + } + } + lines.push(''); + lines.push('Unified verdict guard: compare axes separately; do not blend line and symbol F1.'); + for (const mode of UNIFIED_MODES) { + lines.push( + ` ${mode}: min defined recall=${fmt(unified[mode].minRecall)} FPIS=${unified[mode].fpis} FNIS=${unified[mode].fnis}`, + ); + } + return lines.join('\n'); +} + /** * Plain-language DECISION RECOMMENDATION (F2 — the deliverable that answers * "which is more accurate" as a verdict, not just a table). Derived from the @@ -362,7 +398,7 @@ function renderTable(strata) { * is built to compute) and call-graph's inter-procedural SYMBOL-granularity F1 * (the cross-function reach it is built to compute). */ -function decisionRecommendation(strata, underpowered, exclusions) { +function decisionRecommendation(strata, unified, underpowered, exclusions) { // PDG is precise at intra LINE granularity; callgraph covers inter SYMBOL reach. const pdgIntraF1 = strata.intra.pdg.f1; const pdgIntraP = strata.intra.pdg.precision; @@ -408,6 +444,17 @@ function decisionRecommendation(strata, underpowered, exclusions) { ); } + if (unified) { + lines.push( + `Unified-axis check: current callgraph leaves the intra-line axis empty, and current PDG leaves ` + + `the inter-symbol axis empty. The evaluation-only composed-current baseline combines both current ` + + `outputs and reaches min defined recall=${fmt(unified['composed-current'].minRecall)} with ` + + `FPIS=${unified['composed-current'].fpis} and FNIS=${unified['composed-current'].fnis}. ` + + `A future PDG-only/SDG candidate must match or exceed that recall while reducing or ` + + `bounding FPIS before any default switch.`, + ); + } + lines.push( `VERDICT: the two engines answer DIFFERENT questions at DIFFERENT granularities, and NEITHER ` + `dominates. mode:'callgraph' (the default) is the correct engine for the inter-procedural ` + @@ -464,6 +511,7 @@ async function run() { const exclusions = []; const perRunStrata = []; // K runs × { scope: { mode: aggregate } } + const perRunUnified = []; // K runs × { mode: { intraLine, interSymbol } } let perCaseDetail = null; // last run's per-case detail for the report let degradedCheck = null; @@ -476,6 +524,8 @@ async function run() { for (const m of MODES) perScopeMode[s][m] = []; } const detail = []; + const perUnifiedMode = {}; + for (const m of UNIFIED_MODES) perUnifiedMode[m] = []; for (const fx of fixtures) { if (fx.excluded) { @@ -498,6 +548,17 @@ async function run() { const cgScore = scoreCallgraph(fx.gt, cg.keys); // symbol/inter const pdgScore = scorePdg(fx.gt, pdg.keys); // line/intra + const unifiedTruth = unifiedAis(fx.gt); + const cgUnified = callgraphUnifiedCis(fx.gt, cg.keys); + const pdgUnified = pdgUnifiedCis(pdg.keys); + const composedUnified = composeUnifiedCis(cgUnified, pdgUnified); + const unifiedScores = { + callgraph: scoreUnifiedAxes(cgUnified, unifiedTruth), + pdg: scoreUnifiedAxes(pdgUnified, unifiedTruth), + 'composed-current': scoreUnifiedAxes(composedUnified, unifiedTruth), + }; + for (const m of UNIFIED_MODES) perUnifiedMode[m].push(unifiedScores[m]); + const locusScope = fx.gt.locus; // the stratum this fixture belongs to // A fixture is scored in its OWN locus stratum (intra/inter/mixed), // each mode against its native ground truth (symbol vs line). @@ -526,6 +587,7 @@ async function run() { lines: [...pdg.keys].sort(), score: pdgScore, // vs intra_AIS (line) }, + unified: unifiedScores, }); } } finally { @@ -535,13 +597,16 @@ async function run() { } } - // Aggregate this run's strata. + // Aggregate this run's native strata and unified two-axis comparison. const strata = {}; for (const s of SCOPES) { strata[s] = {}; for (const m of MODES) strata[s][m] = aggregate(perScopeMode[s][m]); } + const unified = {}; + for (const m of UNIFIED_MODES) unified[m] = aggregateUnifiedScores(perUnifiedMode[m]); perRunStrata.push(strata); + perRunUnified.push(unified); if (runIdx === 0) perCaseDetail = detail; } @@ -592,6 +657,33 @@ async function run() { } } + const unified0 = perRunUnified[0]; + const unifiedReport = {}; + for (const mode of UNIFIED_MODES) { + unifiedReport[mode] = { ...unified0[mode] }; + for (const axis of ['intraLine', 'interSymbol']) { + const f1s = perRunUnified + .map((r) => r[mode][axis].f1) + .filter((v) => v !== null && v !== undefined); + const pmeds = perRunUnified + .map((r) => r[mode][axis].precision) + .filter((v) => v !== null && v !== undefined); + const rmeds = perRunUnified + .map((r) => r[mode][axis].recall) + .filter((v) => v !== null && v !== undefined); + unifiedReport[mode][axis] = { + ...unified0[mode][axis], + f1: f1s.length ? median(f1s) : null, + precision: pmeds.length ? median(pmeds) : null, + recall: rmeds.length ? median(rmeds) : null, + }; + } + const minRecalls = perRunUnified + .map((r) => r[mode].minRecall) + .filter((v) => v !== null && v !== undefined); + unifiedReport[mode].minRecall = minRecalls.length ? median(minRecalls) : null; + } + // Underpowered floor (F3): measured cases per stratum after exclusions. const measurableTotal = SCOPES.reduce( (a, s) => a + Math.max(report[s].callgraph.nCases, report[s].pdg.nCases), @@ -612,6 +704,7 @@ async function run() { underpowered, floor: { perStratum: FLOOR_PER_STRATUM, total: FLOOR_TOTAL }, strata: report, + unified: unifiedReport, perCase: perCaseDetail, degradedCheck, annotationFingerprint, @@ -634,6 +727,9 @@ async function run() { ); out.push(renderTable(report)); out.push(''); + out.push('Unified impact axes (additive; native table above is unchanged):'); + out.push(renderUnifiedTable(unifiedReport)); + out.push(''); out.push( 'Per-case: PDG slice (line/intra) and callgraph reach (symbol/inter), with FPIS/FNIS:', ); @@ -669,7 +765,7 @@ async function run() { } out.push(`Annotation fingerprint: ${annotationFingerprint}`); out.push(''); - out.push(decisionRecommendation(report, underpowered, exclusions)); + out.push(decisionRecommendation(report, unifiedReport, underpowered, exclusions)); process.stdout.write(out.join('\n') + '\n'); } diff --git a/gitnexus/bench/impact-pdg/metrics.mjs b/gitnexus/bench/impact-pdg/metrics.mjs index e3c357247..0608d99d5 100644 --- a/gitnexus/bench/impact-pdg/metrics.mjs +++ b/gitnexus/bench/impact-pdg/metrics.mjs @@ -57,6 +57,19 @@ export function lineKey(filePath, line) { return `${filePath}:${line}`; } +/** + * Unified-impact key spaces keep the two granularities explicit. A tagged key + * is never compared across axes: `statement:src/a.ts:10` and + * `symbol:handler@src/a.ts` are different measurement units by design. + */ +export function unifiedLineKey(filePath, line) { + return `statement:${lineKey(filePath, line)}`; +} + +export function unifiedSymbolKey(symbol, filePath) { + return `symbol:${symbolKey(symbol, filePath)}`; +} + /** CIS_pdg = the set of affected-statement LINE keys from an impact pdg result. */ export function pdgLineCis(affectedStatements) { const out = new Set(); @@ -79,6 +92,25 @@ export function intraLineAis(gt) { return out; } +/** Unified AIS = two explicit axes, never one blended line+symbol set. */ +export function unifiedAis(gt) { + const intraLine = new Set(); + for (const e of gt.intra_AIS ?? []) { + if (e && typeof e.line === 'number' && typeof e.filePath === 'string') { + intraLine.add(unifiedLineKey(e.filePath, e.line)); + } + } + + const interSymbol = new Set(); + for (const e of gt.inter_AIS ?? []) { + if (e && typeof e.symbol === 'string' && typeof e.filePath === 'string') { + interSymbol.add(unifiedSymbolKey(e.symbol, e.filePath)); + } + } + + return { intraLine, interSymbol }; +} + /** Canonicalize an iterable of {symbol,filePath} (or pre-made keys) → a Set. */ export function toKeySet(entries) { const out = new Set(); @@ -239,6 +271,58 @@ export function aisByScope(gt) { return { criterionKey: critKey, intra, inter, mixed: new Set([...intra, ...inter]) }; } +const tagSymbolKeys = (keys) => new Set([...keys].map((k) => `symbol:${k}`)); +const tagLineKeys = (keys) => new Set([...keys].map((k) => `statement:${k}`)); + +/** + * Current callgraph unified CIS: inter-symbol axis only. The seed/criterion + * symbol is filtered because `inter_AIS` is cross-function by construction. + */ +export function callgraphUnifiedCis(gt, symbolCisKeys) { + const { criterionKey } = aisByScope(gt); + const inter = new Set([...symbolCisKeys].filter((k) => k !== criterionKey)); + return { intraLine: new Set(), interSymbol: tagSymbolKeys(inter) }; +} + +/** Current PDG unified CIS: intra-line axis only. */ +export function pdgUnifiedCis(lineCisKeys) { + return { intraLine: tagLineKeys(lineCisKeys), interSymbol: new Set() }; +} + +/** Evaluation-only composed baseline: callgraph inter-symbol + PDG intra-line. */ +export function composeUnifiedCis(...parts) { + const intraLine = new Set(); + const interSymbol = new Set(); + for (const part of parts) { + for (const k of part.intraLine ?? []) intraLine.add(k); + for (const k of part.interSymbol ?? []) interSymbol.add(k); + } + return { intraLine, interSymbol }; +} + +/** Score one engine/candidate against unified AIS without blending axes. */ +export function scoreUnifiedAxes(cis, ais) { + return { + intraLine: score(cis.intraLine ?? new Set(), ais.intraLine ?? new Set()), + interSymbol: score(cis.interSymbol ?? new Set(), ais.interSymbol ?? new Set()), + }; +} + +export function aggregateUnifiedScores(perCaseScores) { + const intraLine = aggregate(perCaseScores.map((s) => s.intraLine)); + const interSymbol = aggregate(perCaseScores.map((s) => s.interSymbol)); + const definedRecalls = [intraLine.recall, interSymbol.recall].filter( + (v) => v !== null && v !== undefined, + ); + return { + intraLine, + interSymbol, + minRecall: definedRecalls.length ? Math.min(...definedRecalls) : null, + fpis: (intraLine.fpis ?? 0) + (interSymbol.fpis ?? 0), + fnis: (intraLine.fnis ?? 0) + (interSymbol.fnis ?? 0), + }; +} + /** * Order-independent annotation-set fingerprint (KTD10). Mirrors the * bench/cfg/measure.mjs canonicalization TECHNIQUE (sort every collection, diff --git a/gitnexus/test/unit/impact-pdg-metric-math.test.ts b/gitnexus/test/unit/impact-pdg-metric-math.test.ts index f9d1e85f4..8f809c01f 100644 --- a/gitnexus/test/unit/impact-pdg-metric-math.test.ts +++ b/gitnexus/test/unit/impact-pdg-metric-math.test.ts @@ -213,6 +213,102 @@ describe('impact-pdg metric math — partitionCisByScope() / aisByScope()', () = }); }); +describe('impact-pdg metric math — unified axes', () => { + const gt = { + criterion: { name: 'route', filePath: 'src/mixed.ts', direction: 'downstream' }, + intra_AIS: [ + { symbol: 'route', filePath: 'src/mixed.ts', line: 16 }, + { symbol: 'route', filePath: 'src/mixed.ts', line: 18 }, + ], + inter_AIS: [ + { symbol: 'fast', filePath: 'src/mixed.ts' }, + { symbol: 'slow', filePath: 'src/mixed.ts' }, + ], + }; + + it('builds tagged unified AIS without mixing line and symbol keys', () => { + const ais = M.unifiedAis(gt); + expect([...ais.intraLine].sort()).toEqual([ + 'statement:src/mixed.ts:16', + 'statement:src/mixed.ts:18', + ]); + expect([...ais.interSymbol].sort()).toEqual([ + 'symbol:fast@src/mixed.ts', + 'symbol:slow@src/mixed.ts', + ]); + }); + + it('adapts current engines onto separate unified axes', () => { + const cg = M.callgraphUnifiedCis( + gt, + M.toKeySet([ + M.symbolKey('route', 'src/mixed.ts'), + M.symbolKey('fast', 'src/mixed.ts'), + M.symbolKey('slow', 'src/mixed.ts'), + ]), + ); + const pdg = M.pdgUnifiedCis( + M.pdgLineCis([ + { line: 16, filePath: 'src/mixed.ts' }, + { line: 18, filePath: 'src/mixed.ts' }, + ]), + ); + + expect([...cg.intraLine]).toEqual([]); + expect([...cg.interSymbol].sort()).toEqual([ + 'symbol:fast@src/mixed.ts', + 'symbol:slow@src/mixed.ts', + ]); + expect([...pdg.intraLine].sort()).toEqual([ + 'statement:src/mixed.ts:16', + 'statement:src/mixed.ts:18', + ]); + expect([...pdg.interSymbol]).toEqual([]); + }); + + it('scores composed-current as exact on both axes without a blended F1', () => { + const ais = M.unifiedAis(gt); + const cg = M.callgraphUnifiedCis( + gt, + M.toKeySet([M.symbolKey('fast', 'src/mixed.ts'), M.symbolKey('slow', 'src/mixed.ts')]), + ); + const pdg = M.pdgUnifiedCis( + M.pdgLineCis([ + { line: 16, filePath: 'src/mixed.ts' }, + { line: 18, filePath: 'src/mixed.ts' }, + ]), + ); + + const composed = M.composeUnifiedCis(cg, pdg); + const scored = M.scoreUnifiedAxes(composed, ais); + + expect(scored.intraLine.f1).toBe(1); + expect(scored.interSymbol.f1).toBe(1); + + const agg = M.aggregateUnifiedScores([scored]); + expect(agg.intraLine.f1).toBe(1); + expect(agg.interSymbol.f1).toBe(1); + expect(agg.minRecall).toBe(1); + expect(agg.fpis).toBe(0); + expect(agg.fnis).toBe(0); + expect(agg).not.toHaveProperty('f1'); + }); + + it('makes current standalone engines visibly incomplete on one unified axis', () => { + const ais = M.unifiedAis(gt); + const cg = M.callgraphUnifiedCis(gt, M.toKeySet([M.symbolKey('fast', 'src/mixed.ts')])); + const pdg = M.pdgUnifiedCis(M.pdgLineCis([{ line: 16, filePath: 'src/mixed.ts' }])); + + const cgScore = M.scoreUnifiedAxes(cg, ais); + const pdgScore = M.scoreUnifiedAxes(pdg, ais); + + expect(cgScore.intraLine.recall).toBe(0); + expect(cgScore.interSymbol.recall).toBe(0.5); + expect(pdgScore.intraLine.recall).toBe(0.5); + expect(pdgScore.interSymbol.recall).toBe(0); + }); +}); + describe('impact-pdg metric math — aggregate()', () => { it('macro-averages defined metrics, EXCLUDING nulls (not folding as 0)', () => { const per = [