feat(impact): add unified accuracy axes

This commit is contained in:
Gergo Magyar 2026-06-18 07:31:26 +00:00
parent 84612056cc
commit 4c84ec57c9
4 changed files with 320 additions and 9 deletions

View file

@ -7,9 +7,13 @@
> to run*). The harness drives both `impact` engines over the fixtures — PDG
> **seeded on the criterion's statement line** so it returns the dependence slice
> — prints a stratified P/R/F1 table + a plain-language decision recommendation,
> and gates regressions with `--check`. The measured result: **PDG is exact at
> intra-procedural statement granularity; call-graph is exact at inter-procedural
> symbol granularity; the two answer different questions and neither dominates.**
> and gates regressions with `--check`. It now also prints an additive **unified
> impact axes** table that keeps line-level and symbol-level truth separate while
> comparing current `callgraph`, current `pdg`, and the evaluation-only
> `composed-current` baseline. The measured native result remains: **PDG is exact
> at intra-procedural statement granularity; call-graph is exact at
> inter-procedural symbol granularity; the two answer different questions and
> neither dominates.**
## What this measures
@ -32,6 +36,33 @@ granularity against its native ground truth and reports both side by side. The
single blended number — and the answer is *they answer different questions;
neither strictly dominates*.
## Unified impact axes
The harness also reports a separate unified comparison that is designed for the
next architecture question: *could a future PDG-only / SDG-like impact engine
replace the composition of today's engines?* This report is additive. It does not
replace the native table above, and it does not change `baselines.json` gating.
Unified AIS has two namespaces:
- `statement:<filePath>:<line>` for intra-procedural line truth from `intra_AIS`
- `symbol:<symbol>@<filePath>` for inter-procedural symbol truth from `inter_AIS`
Each engine is adapted onto those axes without lossy projection:
- `callgraph` contributes only the `symbol` axis.
- `pdg` contributes only the `statement` axis.
- `composed-current` is an evaluation-only control row that unions current
callgraph symbols with current PDG statements.
The report intentionally has no single blended unified F1. A future
`pdg-interproc` or SDG candidate must be judged axis-by-axis against
`composed-current` so line precision cannot hide inter-symbol misses, and
symbol recall cannot hide statement-level blindness. The control row is a recall
baseline, not a perfection claim: current PDG can still contribute intra-line
noise on pure-inter fixtures, so a future SDG candidate should match or exceed
recall while reducing or bounding FPIS.
> **A note on `line`.** A whole-symbol PDG slice (no `line`) is empty by design:
> intra-procedural dependence stays inside the function, so every reachable block
> is already part of the whole-symbol seed. The useful PDG mode is the
@ -299,9 +330,13 @@ Read it honestly:
> **cannot answer at all** (it has no notion of a statement).
>
> They **compose**: a full mixed-locus blast radius is the *union* of
> call-graph's inter-symbol reach and PDG's intra-statement slice. Reach for the
> line-seeded PDG when you need statement-level dependence *inside* a function;
> reach for call-graph when you need *cross-function* reach. The earlier verdict
> call-graph's inter-symbol reach and PDG's intra-statement slice. The unified
> axes table makes that composition explicit through the `composed-current` row,
> which is the recall baseline a future SDG / `pdg-interproc` candidate must
> match or exceed while reducing or bounding FPIS. Reach for the line-seeded
> PDG when you need statement-level dependence *inside* a function; reach for
> call-graph when you need
> *cross-function* reach. The earlier verdict
> ("PDG is empty / call-graph wins") was an artifact of the **whole-symbol** seed
> — a whole-symbol slice has nothing to report because intra-procedural dependence
> never leaves the function. Seeding the changed *statement* is what makes PDG's

View file

@ -62,6 +62,12 @@ import {
score,
aggregate,
aisByScope,
unifiedAis,
callgraphUnifiedCis,
pdgUnifiedCis,
composeUnifiedCis,
scoreUnifiedAxes,
aggregateUnifiedScores,
fingerprintAnnotationSet,
median,
} from './metrics.mjs';
@ -74,6 +80,7 @@ const CLI_ENTRY = path.join(REPO_ROOT, 'src', 'cli', 'index.ts');
const SCOPES = ['intra', 'inter', 'mixed'];
const MODES = ['callgraph', 'pdg'];
const UNIFIED_MODES = ['callgraph', 'pdg', 'composed-current'];
// ── F3 minimum-corpus floor (KTD9): below this the harness reports DIRECTION
// only, never a headline decimal verdict. Mirrors the U6 schema test's floor.
@ -355,6 +362,35 @@ function renderTable(strata) {
return lines.join('\n');
}
function renderUnifiedTable(unified) {
const head =
`${pad('Mode', 17)} ${pad('Axis', 13)} ${lpad('P', 7)} ${lpad('R', 7)} ${lpad('F1', 7)} ` +
`${lpad('|CIS|/|AIS|', 11)} ${lpad('FPIS', 6)} ${lpad('FNIS', 6)} ${lpad('n', 4)}`;
const lines = [head, '-'.repeat(head.length)];
const axes = [
['intraLine', 'line/intra'],
['interSymbol', 'symbol/inter'],
];
for (const mode of UNIFIED_MODES) {
for (const [axis, label] of axes) {
const a = unified[mode][axis];
lines.push(
`${pad(mode, 17)} ${pad(label, 13)} ${lpad(fmt(a.precision), 7)} ${lpad(fmt(a.recall), 7)} ` +
`${lpad(fmt(a.f1), 7)} ${lpad(fmt(a.cisAisRatio), 11)} ${lpad(a.fpis, 6)} ${lpad(a.fnis, 6)} ` +
`${lpad(a.nCases, 4)}`,
);
}
}
lines.push('');
lines.push('Unified verdict guard: compare axes separately; do not blend line and symbol F1.');
for (const mode of UNIFIED_MODES) {
lines.push(
` ${mode}: min defined recall=${fmt(unified[mode].minRecall)} FPIS=${unified[mode].fpis} FNIS=${unified[mode].fnis}`,
);
}
return lines.join('\n');
}
/**
* Plain-language DECISION RECOMMENDATION (F2 — the deliverable that answers
* "which is more accurate" as a verdict, not just a table). Derived from the
@ -362,7 +398,7 @@ function renderTable(strata) {
* is built to compute) and call-graph's inter-procedural SYMBOL-granularity F1
* (the cross-function reach it is built to compute).
*/
function decisionRecommendation(strata, underpowered, exclusions) {
function decisionRecommendation(strata, unified, underpowered, exclusions) {
// PDG is precise at intra LINE granularity; callgraph covers inter SYMBOL reach.
const pdgIntraF1 = strata.intra.pdg.f1;
const pdgIntraP = strata.intra.pdg.precision;
@ -408,6 +444,17 @@ function decisionRecommendation(strata, underpowered, exclusions) {
);
}
if (unified) {
lines.push(
`Unified-axis check: current callgraph leaves the intra-line axis empty, and current PDG leaves ` +
`the inter-symbol axis empty. The evaluation-only composed-current baseline combines both current ` +
`outputs and reaches min defined recall=${fmt(unified['composed-current'].minRecall)} with ` +
`FPIS=${unified['composed-current'].fpis} and FNIS=${unified['composed-current'].fnis}. ` +
`A future PDG-only/SDG candidate must match or exceed that recall while reducing or ` +
`bounding FPIS before any default switch.`,
);
}
lines.push(
`VERDICT: the two engines answer DIFFERENT questions at DIFFERENT granularities, and NEITHER ` +
`dominates. mode:'callgraph' (the default) is the correct engine for the inter-procedural ` +
@ -464,6 +511,7 @@ async function run() {
const exclusions = [];
const perRunStrata = []; // K runs × { scope: { mode: aggregate } }
const perRunUnified = []; // K runs × { mode: { intraLine, interSymbol } }
let perCaseDetail = null; // last run's per-case detail for the report
let degradedCheck = null;
@ -476,6 +524,8 @@ async function run() {
for (const m of MODES) perScopeMode[s][m] = [];
}
const detail = [];
const perUnifiedMode = {};
for (const m of UNIFIED_MODES) perUnifiedMode[m] = [];
for (const fx of fixtures) {
if (fx.excluded) {
@ -498,6 +548,17 @@ async function run() {
const cgScore = scoreCallgraph(fx.gt, cg.keys); // symbol/inter
const pdgScore = scorePdg(fx.gt, pdg.keys); // line/intra
const unifiedTruth = unifiedAis(fx.gt);
const cgUnified = callgraphUnifiedCis(fx.gt, cg.keys);
const pdgUnified = pdgUnifiedCis(pdg.keys);
const composedUnified = composeUnifiedCis(cgUnified, pdgUnified);
const unifiedScores = {
callgraph: scoreUnifiedAxes(cgUnified, unifiedTruth),
pdg: scoreUnifiedAxes(pdgUnified, unifiedTruth),
'composed-current': scoreUnifiedAxes(composedUnified, unifiedTruth),
};
for (const m of UNIFIED_MODES) perUnifiedMode[m].push(unifiedScores[m]);
const locusScope = fx.gt.locus; // the stratum this fixture belongs to
// A fixture is scored in its OWN locus stratum (intra/inter/mixed),
// each mode against its native ground truth (symbol vs line).
@ -526,6 +587,7 @@ async function run() {
lines: [...pdg.keys].sort(),
score: pdgScore, // vs intra_AIS (line)
},
unified: unifiedScores,
});
}
} finally {
@ -535,13 +597,16 @@ async function run() {
}
}
// Aggregate this run's strata.
// Aggregate this run's native strata and unified two-axis comparison.
const strata = {};
for (const s of SCOPES) {
strata[s] = {};
for (const m of MODES) strata[s][m] = aggregate(perScopeMode[s][m]);
}
const unified = {};
for (const m of UNIFIED_MODES) unified[m] = aggregateUnifiedScores(perUnifiedMode[m]);
perRunStrata.push(strata);
perRunUnified.push(unified);
if (runIdx === 0) perCaseDetail = detail;
}
@ -592,6 +657,33 @@ async function run() {
}
}
const unified0 = perRunUnified[0];
const unifiedReport = {};
for (const mode of UNIFIED_MODES) {
unifiedReport[mode] = { ...unified0[mode] };
for (const axis of ['intraLine', 'interSymbol']) {
const f1s = perRunUnified
.map((r) => r[mode][axis].f1)
.filter((v) => v !== null && v !== undefined);
const pmeds = perRunUnified
.map((r) => r[mode][axis].precision)
.filter((v) => v !== null && v !== undefined);
const rmeds = perRunUnified
.map((r) => r[mode][axis].recall)
.filter((v) => v !== null && v !== undefined);
unifiedReport[mode][axis] = {
...unified0[mode][axis],
f1: f1s.length ? median(f1s) : null,
precision: pmeds.length ? median(pmeds) : null,
recall: rmeds.length ? median(rmeds) : null,
};
}
const minRecalls = perRunUnified
.map((r) => r[mode].minRecall)
.filter((v) => v !== null && v !== undefined);
unifiedReport[mode].minRecall = minRecalls.length ? median(minRecalls) : null;
}
// Underpowered floor (F3): measured cases per stratum after exclusions.
const measurableTotal = SCOPES.reduce(
(a, s) => a + Math.max(report[s].callgraph.nCases, report[s].pdg.nCases),
@ -612,6 +704,7 @@ async function run() {
underpowered,
floor: { perStratum: FLOOR_PER_STRATUM, total: FLOOR_TOTAL },
strata: report,
unified: unifiedReport,
perCase: perCaseDetail,
degradedCheck,
annotationFingerprint,
@ -634,6 +727,9 @@ async function run() {
);
out.push(renderTable(report));
out.push('');
out.push('Unified impact axes (additive; native table above is unchanged):');
out.push(renderUnifiedTable(unifiedReport));
out.push('');
out.push(
'Per-case: PDG slice (line/intra) and callgraph reach (symbol/inter), with FPIS/FNIS:',
);
@ -669,7 +765,7 @@ async function run() {
}
out.push(`Annotation fingerprint: ${annotationFingerprint}`);
out.push('');
out.push(decisionRecommendation(report, underpowered, exclusions));
out.push(decisionRecommendation(report, unifiedReport, underpowered, exclusions));
process.stdout.write(out.join('\n') + '\n');
}

View file

@ -57,6 +57,19 @@ export function lineKey(filePath, line) {
return `${filePath}:${line}`;
}
/**
* Unified-impact key spaces keep the two granularities explicit. A tagged key
* is never compared across axes: `statement:src/a.ts:10` and
* `symbol:handler@src/a.ts` are different measurement units by design.
*/
export function unifiedLineKey(filePath, line) {
return `statement:${lineKey(filePath, line)}`;
}
export function unifiedSymbolKey(symbol, filePath) {
return `symbol:${symbolKey(symbol, filePath)}`;
}
/** CIS_pdg = the set of affected-statement LINE keys from an impact pdg result. */
export function pdgLineCis(affectedStatements) {
const out = new Set();
@ -79,6 +92,25 @@ export function intraLineAis(gt) {
return out;
}
/** Unified AIS = two explicit axes, never one blended line+symbol set. */
export function unifiedAis(gt) {
const intraLine = new Set();
for (const e of gt.intra_AIS ?? []) {
if (e && typeof e.line === 'number' && typeof e.filePath === 'string') {
intraLine.add(unifiedLineKey(e.filePath, e.line));
}
}
const interSymbol = new Set();
for (const e of gt.inter_AIS ?? []) {
if (e && typeof e.symbol === 'string' && typeof e.filePath === 'string') {
interSymbol.add(unifiedSymbolKey(e.symbol, e.filePath));
}
}
return { intraLine, interSymbol };
}
/** Canonicalize an iterable of {symbol,filePath} (or pre-made keys) → a Set. */
export function toKeySet(entries) {
const out = new Set();
@ -239,6 +271,58 @@ export function aisByScope(gt) {
return { criterionKey: critKey, intra, inter, mixed: new Set([...intra, ...inter]) };
}
const tagSymbolKeys = (keys) => new Set([...keys].map((k) => `symbol:${k}`));
const tagLineKeys = (keys) => new Set([...keys].map((k) => `statement:${k}`));
/**
* Current callgraph unified CIS: inter-symbol axis only. The seed/criterion
* symbol is filtered because `inter_AIS` is cross-function by construction.
*/
export function callgraphUnifiedCis(gt, symbolCisKeys) {
const { criterionKey } = aisByScope(gt);
const inter = new Set([...symbolCisKeys].filter((k) => k !== criterionKey));
return { intraLine: new Set(), interSymbol: tagSymbolKeys(inter) };
}
/** Current PDG unified CIS: intra-line axis only. */
export function pdgUnifiedCis(lineCisKeys) {
return { intraLine: tagLineKeys(lineCisKeys), interSymbol: new Set() };
}
/** Evaluation-only composed baseline: callgraph inter-symbol + PDG intra-line. */
export function composeUnifiedCis(...parts) {
const intraLine = new Set();
const interSymbol = new Set();
for (const part of parts) {
for (const k of part.intraLine ?? []) intraLine.add(k);
for (const k of part.interSymbol ?? []) interSymbol.add(k);
}
return { intraLine, interSymbol };
}
/** Score one engine/candidate against unified AIS without blending axes. */
export function scoreUnifiedAxes(cis, ais) {
return {
intraLine: score(cis.intraLine ?? new Set(), ais.intraLine ?? new Set()),
interSymbol: score(cis.interSymbol ?? new Set(), ais.interSymbol ?? new Set()),
};
}
export function aggregateUnifiedScores(perCaseScores) {
const intraLine = aggregate(perCaseScores.map((s) => s.intraLine));
const interSymbol = aggregate(perCaseScores.map((s) => s.interSymbol));
const definedRecalls = [intraLine.recall, interSymbol.recall].filter(
(v) => v !== null && v !== undefined,
);
return {
intraLine,
interSymbol,
minRecall: definedRecalls.length ? Math.min(...definedRecalls) : null,
fpis: (intraLine.fpis ?? 0) + (interSymbol.fpis ?? 0),
fnis: (intraLine.fnis ?? 0) + (interSymbol.fnis ?? 0),
};
}
/**
* Order-independent annotation-set fingerprint (KTD10). Mirrors the
* bench/cfg/measure.mjs canonicalization TECHNIQUE (sort every collection,

View file

@ -213,6 +213,102 @@ describe('impact-pdg metric math — partitionCisByScope() / aisByScope()', () =
});
});
describe('impact-pdg metric math — unified axes', () => {
const gt = {
criterion: { name: 'route', filePath: 'src/mixed.ts', direction: 'downstream' },
intra_AIS: [
{ symbol: 'route', filePath: 'src/mixed.ts', line: 16 },
{ symbol: 'route', filePath: 'src/mixed.ts', line: 18 },
],
inter_AIS: [
{ symbol: 'fast', filePath: 'src/mixed.ts' },
{ symbol: 'slow', filePath: 'src/mixed.ts' },
],
};
it('builds tagged unified AIS without mixing line and symbol keys', () => {
const ais = M.unifiedAis(gt);
expect([...ais.intraLine].sort()).toEqual([
'statement:src/mixed.ts:16',
'statement:src/mixed.ts:18',
]);
expect([...ais.interSymbol].sort()).toEqual([
'symbol:fast@src/mixed.ts',
'symbol:slow@src/mixed.ts',
]);
});
it('adapts current engines onto separate unified axes', () => {
const cg = M.callgraphUnifiedCis(
gt,
M.toKeySet([
M.symbolKey('route', 'src/mixed.ts'),
M.symbolKey('fast', 'src/mixed.ts'),
M.symbolKey('slow', 'src/mixed.ts'),
]),
);
const pdg = M.pdgUnifiedCis(
M.pdgLineCis([
{ line: 16, filePath: 'src/mixed.ts' },
{ line: 18, filePath: 'src/mixed.ts' },
]),
);
expect([...cg.intraLine]).toEqual([]);
expect([...cg.interSymbol].sort()).toEqual([
'symbol:fast@src/mixed.ts',
'symbol:slow@src/mixed.ts',
]);
expect([...pdg.intraLine].sort()).toEqual([
'statement:src/mixed.ts:16',
'statement:src/mixed.ts:18',
]);
expect([...pdg.interSymbol]).toEqual([]);
});
it('scores composed-current as exact on both axes without a blended F1', () => {
const ais = M.unifiedAis(gt);
const cg = M.callgraphUnifiedCis(
gt,
M.toKeySet([M.symbolKey('fast', 'src/mixed.ts'), M.symbolKey('slow', 'src/mixed.ts')]),
);
const pdg = M.pdgUnifiedCis(
M.pdgLineCis([
{ line: 16, filePath: 'src/mixed.ts' },
{ line: 18, filePath: 'src/mixed.ts' },
]),
);
const composed = M.composeUnifiedCis(cg, pdg);
const scored = M.scoreUnifiedAxes(composed, ais);
expect(scored.intraLine.f1).toBe(1);
expect(scored.interSymbol.f1).toBe(1);
const agg = M.aggregateUnifiedScores([scored]);
expect(agg.intraLine.f1).toBe(1);
expect(agg.interSymbol.f1).toBe(1);
expect(agg.minRecall).toBe(1);
expect(agg.fpis).toBe(0);
expect(agg.fnis).toBe(0);
expect(agg).not.toHaveProperty('f1');
});
it('makes current standalone engines visibly incomplete on one unified axis', () => {
const ais = M.unifiedAis(gt);
const cg = M.callgraphUnifiedCis(gt, M.toKeySet([M.symbolKey('fast', 'src/mixed.ts')]));
const pdg = M.pdgUnifiedCis(M.pdgLineCis([{ line: 16, filePath: 'src/mixed.ts' }]));
const cgScore = M.scoreUnifiedAxes(cg, ais);
const pdgScore = M.scoreUnifiedAxes(pdg, ais);
expect(cgScore.intraLine.recall).toBe(0);
expect(cgScore.interSymbol.recall).toBe(0.5);
expect(pdgScore.intraLine.recall).toBe(0.5);
expect(pdgScore.interSymbol.recall).toBe(0);
});
});
describe('impact-pdg metric math — aggregate()', () => {
it('macro-averages defined metrics, EXCLUDING nulls (not folding as 0)', () => {
const per = [