From b85c80143387a1aa7d55b1f32babd100fface07e Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Thu, 18 Jun 2026 10:44:30 +0000 Subject: [PATCH] feat(impact): real-code perf/quality probe for pdg impact mode Add bench/impact-pdg/real-code.mjs: a latency + quality-proxy probe that runs both impact engines against an already-indexed real repo (default GitNexus) and checks that unified mode:'pdg' preserves the callgraph inter-procedural symbol set, what it costs, and how honest its PDG evidence/degraded signals are. Unlike measure.mjs (the AIS-backed fixture accuracy gate), a real repo has no curated ground truth, so this reports quality proxies, not accuracy. Each default case is anchored on a CFG block-start line so every case exercises a real intra-procedural slice; a mid-block anchor degrades to pdg-no-block-at-line, which the harness still detects and counts. Measured on the live GitNexus index (PDG layer via analyze --pdg, ~171k BasicBlocks): unified pdg reproduces callgraph symbol reach exactly (recall = precision = 1.0 on all cases), ~1.2-1.4x latency overhead, and downstream statement-anchored seeds correctly label out-of-slice reach unproven-bridge (an expected proven-vs-reachable signal, not a regression). Adds a deterministic helper unit test (pure metric math, no analyze/DB) and README docs covering the probe, its gates, and a representative run. Co-Authored-By: Claude Opus 4.8 (1M context) --- gitnexus/bench/impact-pdg/README.md | 50 ++ gitnexus/bench/impact-pdg/real-code.mjs | 457 ++++++++++++++++++ .../unit/impact-pdg-real-code-metrics.test.ts | 132 +++++ 3 files changed, 639 insertions(+) create mode 100644 gitnexus/bench/impact-pdg/real-code.mjs create mode 100644 gitnexus/test/unit/impact-pdg-real-code-metrics.test.ts diff --git a/gitnexus/bench/impact-pdg/README.md b/gitnexus/bench/impact-pdg/README.md index 135e717da..ce0918eb7 100644 --- a/gitnexus/bench/impact-pdg/README.md +++ b/gitnexus/bench/impact-pdg/README.md @@ -416,8 +416,58 @@ node --import tsx bench/impact-pdg/measure.mjs # print the strat node --import tsx bench/impact-pdg/measure.mjs --json # machine report (for re-baselining) node --import tsx bench/impact-pdg/measure.mjs --check # gate against baselines.json (exit non-zero on regression) node --import tsx bench/impact-pdg/measure.mjs --only=a,b,c # fast subset (substrate smoke) +node --import tsx bench/impact-pdg/real-code.mjs # latency + quality-proxy probe on indexed GitNexus +node --import tsx bench/impact-pdg/real-code.mjs --json --check # machine report + broad real-code gates ``` +### Real-code performance and quality proxy probe + +`real-code.mjs` complements the AIS-backed fixture harness. It runs direct +`LocalBackend.callTool("impact", ...)` calls against an already-indexed real +repository (default `--repo GitNexus`) and measures: + +- callgraph vs PDG median/p95 latency over `--repeat` samples; +- whether unified PDG's inter-procedural symbol reach preserves the callgraph + symbol set for the same target/direction; +- degraded, partial, no-block-at-line, and PDG bridge evidence counts. + +This is a quality proxy, not an accuracy score: a real repo has no curated AIS, +so the probe cannot prove correctness. Use it to catch performance regressions, +degraded indexes, symbol-reach drift, and excessive `unproven-bridge` evidence on +real code. Use `measure.mjs` for the ground-truth precision/recall/F1 gate. + +The default cases are statement-anchored at a CFG **block-start** line. The CFG +coalesces straight-line statements into one `BasicBlock`, so a mid-block anchor +resolves to no block start and degrades to `pdg-no-block-at-line` — honest, but it +then exercises only the symbol axis. The harness still detects and counts that +degradation; the curated anchors avoid it so every case also exercises a real +intra-procedural slice. (This is the same statement-anchoring discipline the +fixture corpus uses, applied to real code.) + +A representative run on the indexed GitNexus tree (~17.5k symbols, PDG layer +persisted via `analyze --pdg` with ~171k `BasicBlock`s) — read it *directionally*, +not as a baseline, since wall-clock latency is host- and noise-dependent: + +- **Symbol reach is preserved exactly.** Unified `mode:'pdg'` reproduces the + `mode:'callgraph'` inter-procedural symbol set on every case — mean and min + recall = precision = **1.000**. This is the load-bearing check: the PDG-facing + result must not silently drop or invent cross-function reach. +- **Each case carries a real statement slice** (`affectedStatements` non-empty, + 2–27 statements here), so the intra axis is genuinely exercised. +- **Latency overhead is modest** — PDG median ≈ **1.2–1.4×** the callgraph median + (callgraph ≈ 90–250 ms/case, PDG ≈ 150–280 ms/case). The first call of a fresh + backend carries a one-time DB-warmup spike the p95 reflects. +- **Bridge evidence is direction-shaped, by design.** Downstream + statement-anchored seeds label most inter-procedural reach `unproven-bridge` + (the symbol's first-hop call site sits in a *different* statement than the + seeded one, so the local slice does not prove the dependence); upstream and + whole-symbol reach is `callgraph-bridge`. So `unprovenBridgeRatio ≈ 0.7` is the + *expected* shape for statement-anchored downstream seeds — a faithful + proven-vs-reachable signal, **not** a regression. +- **No degraded / error / partial / no-block-at-line cases**, and `--check` is + green. Default gates: min symbol recall ≥ 0.95, PDG median ≤ 5000 ms (override + via `GN_REAL_CODE_PDG_MIN_SYMBOL_RECALL` / `GN_REAL_CODE_PDG_MAX_MEDIAN_MS`). + ### Re-baseline (after a reviewed accuracy or ground-truth change) 1. `node --import tsx bench/impact-pdg/measure.mjs --json` and read diff --git a/gitnexus/bench/impact-pdg/real-code.mjs b/gitnexus/bench/impact-pdg/real-code.mjs new file mode 100644 index 000000000..1a888ba7c --- /dev/null +++ b/gitnexus/bench/impact-pdg/real-code.mjs @@ -0,0 +1,457 @@ +/** + * Real-code performance and quality proxy probe for impact modes. + * + * This complements `measure.mjs`, which is the ground-truth accuracy gate over + * curated fixtures. A real repository does not have an AIS annotation set, so + * this probe does NOT claim accuracy. It checks whether unified `mode:'pdg'` + * preserves the established callgraph symbol reach on a real index, how much it + * costs, and how honest its PDG evidence/degraded signals are. + */ +import fs from 'node:fs'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { fileURLToPath } from 'node:url'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const REPO_ROOT = path.resolve(__dirname, '..', '..'); + +export const DEFAULT_REAL_CODE_CASES = [ + { + name: 'cli-format-impact-upstream', + target: 'formatImpactResult', + file_path: 'gitnexus/src/cli/eval-server.ts', + kind: 'Function', + direction: 'upstream', + line: 208, + }, + { + name: 'cli-impact-command-downstream', + target: 'impactCommand', + file_path: 'gitnexus/src/cli/tool.ts', + kind: 'Function', + direction: 'downstream', + // Block-start line of the coalesced backend-call statement group. The CFG + // coalesces lines 162-192 into one BasicBlock, so an anchor mid-block (e.g. + // 173) lands on no block start and degrades to pdg-no-block-at-line. Seed the + // block's start line so the intra slice is exercised on a real statement. + line: 162, + }, + { + name: 'pdg-engine-downstream', + target: 'runImpactPDG', + file_path: 'gitnexus/src/mcp/local/pdg-impact.ts', + kind: 'Function', + direction: 'downstream', + // Block-start line of the function's opening coalesced statement group + // (the destructure + budget setup spanning 912+). Mid-block lines like 952 + // resolve to no block start; 912 seeds a real, statement-rich intra slice. + line: 912, + }, + { + name: 'pdg-dispatch-upstream', + target: '_impactImpl', + file_path: 'gitnexus/src/mcp/local/local-backend.ts', + kind: 'Method', + direction: 'upstream', + line: 4427, + }, + { + name: 'pdg-compose-downstream', + target: 'composeUnifiedPdgImpactResult', + file_path: 'gitnexus/src/mcp/local/local-backend.ts', + kind: 'Method', + direction: 'downstream', + line: 4850, + }, +]; + +function readOption(argv, name, fallback = undefined) { + const eq = argv.find((arg) => arg.startsWith(`--${name}=`)); + if (eq) return eq.slice(name.length + 3); + const idx = argv.indexOf(`--${name}`); + if (idx >= 0 && idx + 1 < argv.length) return argv[idx + 1]; + return fallback; +} + +function hasFlag(argv, name) { + return argv.includes(`--${name}`); +} + +export function median(xs) { + if (xs.length === 0) return null; + const sorted = [...xs].sort((a, b) => a - b); + const mid = Math.floor(sorted.length / 2); + return sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2; +} + +export function percentile(xs, pct) { + if (xs.length === 0) return null; + const sorted = [...xs].sort((a, b) => a - b); + const idx = Math.min(sorted.length - 1, Math.max(0, Math.ceil((pct / 100) * sorted.length) - 1)); + return sorted[idx]; +} + +function round(value, digits = 3) { + if (value === null || value === undefined || Number.isNaN(value)) return null; + const scale = 10 ** digits; + return Math.round(value * scale) / scale; +} + +function fmt(value, digits = 1) { + return value === null || value === undefined ? 'n/a' : Number(value).toFixed(digits); +} + +export function symbolKeysFromByDepth(byDepth) { + const keys = new Set(); + for (const items of Object.values(byDepth ?? {})) { + for (const item of items ?? []) { + if (!item || typeof item !== 'object') continue; + if (typeof item.id === 'string' && item.id.length > 0) { + keys.add(item.id); + continue; + } + const name = typeof item.name === 'string' ? item.name : '(unknown)'; + const filePath = typeof item.filePath === 'string' ? item.filePath : '(unknown)'; + keys.add(`${name}@${filePath}`); + } + } + return keys; +} + +export function compareSymbolSets(reference, candidate) { + const ref = new Set(reference); + const cand = new Set(candidate); + const overlap = [...ref].filter((key) => cand.has(key)); + const referenceOnly = [...ref].filter((key) => !cand.has(key)).sort(); + const candidateOnly = [...cand].filter((key) => !ref.has(key)).sort(); + const unionSize = new Set([...ref, ...cand]).size; + return { + referenceSize: ref.size, + candidateSize: cand.size, + overlapSize: overlap.length, + recallVsReference: ref.size === 0 ? null : overlap.length / ref.size, + precisionVsReference: cand.size === 0 ? null : overlap.length / cand.size, + jaccard: unionSize === 0 ? null : overlap.length / unionSize, + referenceOnly, + candidateOnly, + }; +} + +function sumEvidenceCounts(results) { + const counts = {}; + for (const result of results) { + const evidenceCounts = + result?.pdgInterprocedural?.evidenceCounts ?? + result?.pdgEvidence?.interproceduralEvidenceCounts ?? + {}; + for (const [key, value] of Object.entries(evidenceCounts)) { + counts[key] = (counts[key] ?? 0) + Number(value ?? 0); + } + } + return counts; +} + +function readCases(caseFile) { + if (!caseFile) return DEFAULT_REAL_CODE_CASES; + const resolved = path.resolve(process.cwd(), caseFile); + const parsed = JSON.parse(fs.readFileSync(resolved, 'utf8')); + const cases = Array.isArray(parsed) ? parsed : parsed.cases; + if (!Array.isArray(cases) || cases.length === 0) { + throw new Error(`case file ${resolved} must contain a non-empty array or { "cases": [...] }`); + } + return cases; +} + +async function timedImpact(backend, params) { + const started = performance.now(); + const result = await backend.callTool('impact', params); + return { result, ms: performance.now() - started }; +} + +async function measureCase(backend, testCase, options) { + const baseParams = { + repo: options.repo, + target: testCase.target, + file_path: testCase.file_path, + kind: testCase.kind, + direction: testCase.direction ?? 'upstream', + maxDepth: options.depth, + includeTests: options.includeTests, + limit: options.limit, + }; + const callgraphTimes = []; + const pdgTimes = []; + let callgraphResult = null; + let pdgResult = null; + + for (let i = 0; i < options.repeat; i++) { + const callgraph = await timedImpact(backend, { ...baseParams, mode: 'callgraph' }); + callgraphTimes.push(callgraph.ms); + callgraphResult = callgraph.result; + + const pdg = await timedImpact(backend, { + ...baseParams, + mode: 'pdg', + ...(Number.isInteger(testCase.line) ? { line: testCase.line } : {}), + }); + pdgTimes.push(pdg.ms); + pdgResult = pdg.result; + } + + const callgraphKeys = symbolKeysFromByDepth(callgraphResult?.byDepth ?? {}); + const pdgInterByDepth = + pdgResult?.interproceduralByDepth ?? pdgResult?.pdgInterprocedural?.byDepth ?? {}; + const pdgInterKeys = symbolKeysFromByDepth(pdgInterByDepth); + const symbolAgreement = compareSymbolSets(callgraphKeys, pdgInterKeys); + const evidenceCounts = + pdgResult?.pdgInterprocedural?.evidenceCounts ?? + pdgResult?.pdgEvidence?.interproceduralEvidenceCounts ?? + {}; + + return { + name: testCase.name ?? testCase.target, + target: testCase.target, + filePath: testCase.file_path, + kind: testCase.kind, + direction: baseParams.direction, + line: Number.isInteger(testCase.line) ? testCase.line : null, + latencyMs: { + callgraph: { + median: round(median(callgraphTimes)), + p95: round(percentile(callgraphTimes, 95)), + samples: callgraphTimes.map((v) => round(v)), + }, + pdg: { + median: round(median(pdgTimes)), + p95: round(percentile(pdgTimes, 95)), + samples: pdgTimes.map((v) => round(v)), + }, + pdgOverCallgraphMedian: + median(callgraphTimes) && median(callgraphTimes) > 0 + ? round(median(pdgTimes) / median(callgraphTimes)) + : null, + }, + callgraph: { + error: callgraphResult?.error ?? null, + impactedCount: callgraphResult?.impactedCount ?? 0, + risk: callgraphResult?.risk ?? null, + epistemic: callgraphResult?.epistemic ?? null, + partial: Boolean(callgraphResult?.partial), + symbolCount: callgraphKeys.size, + }, + pdg: { + error: pdgResult?.error ?? null, + pdgLayer: pdgResult?.pdgLayer ?? 'ready', + epistemic: pdgResult?.epistemic ?? null, + partial: Boolean(pdgResult?.partial || pdgResult?.pdgInterprocedural?.partial), + impactedCount: pdgResult?.impactedCount ?? 0, + affectedStatementCount: pdgResult?.affectedStatementCount ?? 0, + blockCount: pdgResult?.blockCount ?? 0, + interproceduralSymbolCount: pdgInterKeys.size, + evidence: pdgResult?.pdgInterprocedural?.evidence ?? pdgResult?.pdgEvidence?.interprocedural, + evidenceCounts, + }, + symbolAgreement, + }; +} + +export function summarizeCases(cases) { + const ratios = cases + .map((c) => c.latencyMs.pdgOverCallgraphMedian) + .filter((v) => v !== null && v !== undefined); + const callgraphMedians = cases + .map((c) => c.latencyMs.callgraph.median) + .filter((v) => v !== null && v !== undefined); + const pdgMedians = cases + .map((c) => c.latencyMs.pdg.median) + .filter((v) => v !== null && v !== undefined); + const comparable = cases.filter((c) => c.symbolAgreement.recallVsReference !== null); + const recalls = comparable.map((c) => c.symbolAgreement.recallVsReference); + const precisions = comparable + .map((c) => c.symbolAgreement.precisionVsReference) + .filter((v) => v !== null && v !== undefined); + const degradedCases = cases.filter((c) => c.pdg.pdgLayer !== 'ready'); + const errorCases = cases.filter((c) => c.callgraph.error || c.pdg.error); + const partialCases = cases.filter((c) => c.callgraph.partial || c.pdg.partial); + const noBlockAtLineCases = cases.filter((c) => c.pdg.epistemic === 'pdg-no-block-at-line'); + const evidenceCounts = sumEvidenceCounts(cases.map((c) => ({ pdgInterprocedural: c.pdg }))); + const totalBridgeSymbols = Object.values(evidenceCounts).reduce((a, b) => a + Number(b ?? 0), 0); + + return { + performance: { + callgraphMedianMs: round(median(callgraphMedians)), + pdgMedianMs: round(median(pdgMedians)), + pdgP95Ms: round(percentile(cases.map((c) => c.latencyMs.pdg.p95).filter(Boolean), 95)), + pdgOverCallgraphMedian: round(median(ratios)), + }, + qualityProxy: { + comparableCases: comparable.length, + meanSymbolRecallVsCallgraph: recalls.length + ? round(recalls.reduce((a, b) => a + b, 0) / recalls.length) + : null, + minSymbolRecallVsCallgraph: recalls.length ? round(Math.min(...recalls)) : null, + meanSymbolPrecisionVsCallgraph: precisions.length + ? round(precisions.reduce((a, b) => a + b, 0) / precisions.length) + : null, + degradedCaseCount: degradedCases.length, + errorCaseCount: errorCases.length, + partialCaseCount: partialCases.length, + noBlockAtLineCaseCount: noBlockAtLineCases.length, + evidenceCounts, + unprovenBridgeRatio: + totalBridgeSymbols > 0 + ? round((evidenceCounts['unproven-bridge'] ?? 0) / totalBridgeSymbols) + : null, + }, + }; +} + +export function evaluateCheckGates(report, env = process.env) { + const failures = []; + const minRecall = Number(env.GN_REAL_CODE_PDG_MIN_SYMBOL_RECALL ?? 0.95); + const maxMedianMs = Number(env.GN_REAL_CODE_PDG_MAX_MEDIAN_MS ?? 5000); + const quality = report.summary.qualityProxy; + const perf = report.summary.performance; + + if (report.cases.length === 0) failures.push('no real-code cases were measured'); + if (quality.errorCaseCount > 0) + failures.push(`${quality.errorCaseCount} case(s) returned errors`); + if (quality.degradedCaseCount > 0) { + failures.push(`${quality.degradedCaseCount} case(s) reported a degraded PDG layer`); + } + if ( + quality.minSymbolRecallVsCallgraph !== null && + quality.minSymbolRecallVsCallgraph < minRecall + ) { + failures.push( + `min PDG symbol recall vs callgraph ${quality.minSymbolRecallVsCallgraph} < ${minRecall}`, + ); + } + if (perf.pdgMedianMs !== null && perf.pdgMedianMs > maxMedianMs) { + failures.push(`PDG median latency ${perf.pdgMedianMs}ms > ${maxMedianMs}ms`); + } + return failures; +} + +function renderText(report, failures) { + const lines = []; + const perf = report.summary.performance; + const quality = report.summary.qualityProxy; + lines.push('=== impact-PDG real-code performance/quality probe ==='); + lines.push( + `repo ${report.repo} | cases ${report.cases.length} | repeat ${report.repeat} | includeTests=${report.includeTests}`, + ); + lines.push(''); + lines.push( + `Latency: callgraph median ${fmt(perf.callgraphMedianMs)}ms, ` + + `pdg median ${fmt(perf.pdgMedianMs)}ms, pdg p95 ${fmt(perf.pdgP95Ms)}ms, ` + + `median overhead ${fmt(perf.pdgOverCallgraphMedian, 2)}x`, + ); + lines.push( + `Quality proxy: min PDG symbol recall vs callgraph ${fmt( + quality.minSymbolRecallVsCallgraph, + 3, + )}, mean recall ${fmt(quality.meanSymbolRecallVsCallgraph, 3)}, ` + + `mean precision ${fmt(quality.meanSymbolPrecisionVsCallgraph, 3)}`, + ); + lines.push( + `Signals: degraded=${quality.degradedCaseCount}, errors=${quality.errorCaseCount}, ` + + `partial=${quality.partialCaseCount}, no-block-at-line=${quality.noBlockAtLineCaseCount}, ` + + `unprovenBridgeRatio=${fmt(quality.unprovenBridgeRatio, 3)}`, + ); + lines.push(`Evidence counts: ${JSON.stringify(quality.evidenceCounts)}`); + lines.push(''); + lines.push('Per case:'); + for (const c of report.cases) { + lines.push( + ` ${c.name}: cg ${fmt(c.latencyMs.callgraph.median)}ms/${c.callgraph.symbolCount} symbols, ` + + `pdg ${fmt(c.latencyMs.pdg.median)}ms/${c.pdg.interproceduralSymbolCount} inter-symbols, ` + + `statements=${c.pdg.affectedStatementCount}, recall=${fmt( + c.symbolAgreement.recallVsReference, + 3, + )}, precision=${fmt(c.symbolAgreement.precisionVsReference, 3)}, ` + + `evidence=${c.pdg.evidence ?? 'n/a'}`, + ); + if (c.callgraph.error || c.pdg.error || c.pdg.pdgLayer !== 'ready') { + lines.push( + ` status: callgraphError=${c.callgraph.error ?? 'none'} pdgError=${ + c.pdg.error ?? 'none' + } pdgLayer=${c.pdg.pdgLayer}`, + ); + } + if (c.symbolAgreement.referenceOnly.length > 0) { + lines.push( + ` callgraph-only symbols: ${c.symbolAgreement.referenceOnly.slice(0, 5).join(', ')}`, + ); + } + } + lines.push(''); + lines.push( + 'Interpretation: this real-code probe measures latency and quality proxies, not accuracy. ' + + 'The curated fixture harness remains the AIS-backed accuracy gate.', + ); + if (failures.length > 0) { + lines.push(''); + for (const failure of failures) lines.push(`[impact-pdg-real-code --check] FAIL: ${failure}`); + } + return lines.join('\n'); +} + +async function run() { + const argv = process.argv.slice(2); + const repo = readOption(argv, 'repo', 'GitNexus'); + const repeat = Math.max(1, Number(readOption(argv, 'repeat', '3'))); + const depth = Math.max(1, Number(readOption(argv, 'depth', '3'))); + const limit = Math.max(1, Number(readOption(argv, 'limit', '100'))); + const includeTests = readOption(argv, 'include-tests', 'true') !== 'false'; + const caseFile = readOption(argv, 'case-file'); + const json = hasFlag(argv, 'json'); + const check = hasFlag(argv, 'check'); + const cases = readCases(caseFile); + + const { LocalBackend } = await import( + path.join(REPO_ROOT, 'src', 'mcp', 'local', 'local-backend.ts') + ); + const backend = new LocalBackend(); + const initialized = await backend.init(); + if (!initialized) + throw new Error('no indexed repositories found; run gitnexus analyze --pdg first'); + + try { + const measured = []; + for (const testCase of cases) { + measured.push( + await measureCase(backend, testCase, { repo, repeat, depth, limit, includeTests }), + ); + } + const report = { + repo, + repeat, + depth, + limit, + includeTests, + generatedAt: new Date().toISOString(), + note: 'Real-code probe: latency plus quality proxies only. Accuracy requires AIS-backed fixtures.', + cases: measured, + summary: summarizeCases(measured), + }; + const failures = evaluateCheckGates(report); + + if (json) { + process.stdout.write(JSON.stringify({ ...report, checkFailures: failures }, null, 2) + '\n'); + } else { + process.stdout.write(renderText(report, failures) + '\n'); + } + + if (check && failures.length > 0) process.exit(1); + } finally { + await backend.dispose().catch(() => {}); + } +} + +if (path.resolve(process.argv[1] ?? '') === fileURLToPath(import.meta.url)) { + run().catch((err) => { + process.stderr.write(`[impact-pdg-real-code] ERROR: ${err?.stack || err}\n`); + process.exit(1); + }); +} diff --git a/gitnexus/test/unit/impact-pdg-real-code-metrics.test.ts b/gitnexus/test/unit/impact-pdg-real-code-metrics.test.ts new file mode 100644 index 000000000..cbf3a93db --- /dev/null +++ b/gitnexus/test/unit/impact-pdg-real-code-metrics.test.ts @@ -0,0 +1,132 @@ +import { describe, expect, it } from 'vitest'; + +import { + compareSymbolSets, + evaluateCheckGates, + median, + percentile, + summarizeCases, + symbolKeysFromByDepth, +} from '../../bench/impact-pdg/real-code.mjs'; + +describe('impact-pdg real-code metric helpers', () => { + it('extracts stable symbol keys from byDepth records', () => { + const keys = symbolKeysFromByDepth({ + 1: [ + { id: 'Function:src/a.ts:a', name: 'a', filePath: 'src/a.ts' }, + { id: null, name: 'dynamicTarget', filePath: 'src/b.ts' }, + ], + }); + + expect([...keys].sort()).toEqual(['Function:src/a.ts:a', 'dynamicTarget@src/b.ts']); + }); + + it('compares candidate symbol reach against a reference set', () => { + const comparison = compareSymbolSets(new Set(['A', 'B']), new Set(['A', 'C'])); + + expect(comparison).toMatchObject({ + referenceSize: 2, + candidateSize: 2, + overlapSize: 1, + recallVsReference: 0.5, + precisionVsReference: 0.5, + jaccard: 1 / 3, + referenceOnly: ['B'], + candidateOnly: ['C'], + }); + }); + + it('summarizes latency and quality proxy metrics across cases', () => { + const summary = summarizeCases([ + { + latencyMs: { + callgraph: { median: 10 }, + pdg: { median: 25, p95: 30 }, + pdgOverCallgraphMedian: 2.5, + }, + callgraph: { error: null, partial: false }, + pdg: { + error: null, + pdgLayer: 'ready', + partial: false, + epistemic: 'pdg-intra-procedural', + evidenceCounts: { 'callgraph-bridge': 2 }, + }, + symbolAgreement: { + recallVsReference: 1, + precisionVsReference: 1, + }, + }, + { + latencyMs: { + callgraph: { median: 20 }, + pdg: { median: 40, p95: 45 }, + pdgOverCallgraphMedian: 2, + }, + callgraph: { error: null, partial: false }, + pdg: { + error: null, + pdgLayer: 'ready', + partial: true, + epistemic: 'pdg-intra-procedural', + evidenceCounts: { 'unproven-bridge': 1 }, + }, + symbolAgreement: { + recallVsReference: 0.5, + precisionVsReference: 1, + }, + }, + ] as any); + + expect(summary.performance).toMatchObject({ + callgraphMedianMs: 15, + pdgMedianMs: 32.5, + pdgP95Ms: 45, + pdgOverCallgraphMedian: 2.25, + }); + expect(summary.qualityProxy).toMatchObject({ + comparableCases: 2, + meanSymbolRecallVsCallgraph: 0.75, + minSymbolRecallVsCallgraph: 0.5, + meanSymbolPrecisionVsCallgraph: 1, + degradedCaseCount: 0, + errorCaseCount: 0, + partialCaseCount: 1, + evidenceCounts: { 'callgraph-bridge': 2, 'unproven-bridge': 1 }, + unprovenBridgeRatio: 0.333, + }); + }); + + it('reports explicit check failures for degraded quality and slow PDG medians', () => { + const failures = evaluateCheckGates( + { + cases: [{}], + summary: { + performance: { pdgMedianMs: 600 }, + qualityProxy: { + errorCaseCount: 1, + degradedCaseCount: 1, + minSymbolRecallVsCallgraph: 0.5, + }, + }, + } as any, + { + GN_REAL_CODE_PDG_MIN_SYMBOL_RECALL: '0.9', + GN_REAL_CODE_PDG_MAX_MEDIAN_MS: '500', + } as any, + ); + + expect(failures).toEqual([ + '1 case(s) returned errors', + '1 case(s) reported a degraded PDG layer', + 'min PDG symbol recall vs callgraph 0.5 < 0.9', + 'PDG median latency 600ms > 500ms', + ]); + }); + + it('computes median and percentile deterministically', () => { + expect(median([9, 1, 3])).toBe(3); + expect(median([9, 1, 3, 5])).toBe(4); + expect(percentile([10, 30, 20], 95)).toBe(30); + }); +});