perf(cfg): sparse change-driven reaching-defs solver + canonical truncation (#2201 U3,U4)

This commit is contained in:
Gergo Magyar 2026-06-15 13:44:07 +00:00
parent 0150a44339
commit aa84c91d35
2 changed files with 260 additions and 7 deletions

View file

@ -212,6 +212,21 @@ export function computeReachingDefsDense(
return solveReachingDefs(cfg, limits, computeInSetsDense);
}
/**
* Sparse, change-driven reaching-defs (#2201) — the production solve, exposed
* directly so the equivalence fuzz can gate it against the dense oracle before
* {@link computeReachingDefs} is switched over to it (U5). See
* {@link computeInSetsSparse} for the algorithm and byte-identical contract.
*
* @internal exported only for the equivalence fuzz harness and the cfg bench.
*/
export function computeReachingDefsSparse(
cfg: FunctionCfg,
limits?: ReachingDefsLimits,
): FunctionDefUse {
return solveReachingDefs(cfg, limits, computeInSetsSparse);
}
/**
* Shared orchestrator: the no-facts / overflow guards, the harvest, the
* adjacency build, the swappable IN-set computation, and the statement sweep.
@ -438,6 +453,173 @@ function computeInSetsDense(
return { converged: true, inSets };
}
/**
* SPARSE IN-set computer (#2201) — the production solver. Computes the SAME
* per-block entry reaching lattices as {@link computeInSetsDense}, but via a
* change-driven worklist over (block, binding) PAIRS instead of dense per-block
* lattice merges over a multi-pass fixpoint. Reaching-defs is component-wise
* independent per binding (a binding's transfer never reads another binding's
* set), so the product-lattice least fixed point equals the product of the
* per-binding least fixed points — this solve is provably equal to the dense
* one, and the perf win is that a binding is only ever (re)visited when one of
* its own predecessors' contributions changed (no dense per-block spine copy,
* no re-merge of unrelated bindings, no loop-depth pass multiplier on the
* shallow/loop-local variables that dominate real code).
*
* BYTE-IDENTICAL DISCIPLINE: each visit RECOMPUTES in_v[b] from scratch using
* the exact dense merge order (sorted predecessors, first contributor shared,
* copy-on-extend) — see {@link mergePreds}. Because the merge rebuilds from the
* converged predecessor sets, the final set's INSERTION order is independent of
* worklist visit order and matches the dense solver, which is what makes a
* maxFacts-TRUNCATED result (whose surviving subset depends on the sweep's
* pre-sort emission order) byte-identical too.
*
* BUDGET (KTD5): the dense ceiling counts block dequeues; this counts (block,
* binding) dequeues. The budget is scaled by the binding count so it NEVER
* trips on a function the dense solver computes (no coverage regression) while
* still bounding the one adversarial residual — a single variable threaded
* through every level of a pathologically deep nest, which stays O(depth²) here
* (a documented full-SSA follow-up). On realistic deep nests the change-driven
* solve completes far under budget, so the dense ceiling effectively never
* fires (#2201 acceptance).
*
* @internal
*/
function computeInSetsSparse(
cfg: FunctionCfg,
n: number,
h: Harvest,
adj: Adjacency,
limits: ReachingDefsLimits | undefined,
): InSetsResult {
const { gen, allDefsGen } = h;
const { preds, succs, throwSuccs } = adj;
const nBindings = cfg.bindings?.length ?? 0;
if (nBindings === 0) {
return { converged: true, inSets: new Array(n).fill(EMPTY_LATTICE) };
}
// Per-block IN/OUT lattices, populated incrementally per binding. Sets are
// either nonempty or absent (never empty), mirroring the dense maps so the
// sweep's `.get(u)` sees identical undefined-vs-set results.
const inSets: Lattice[] = Array.from({ length: n }, () => new Map());
const outSets: Lattice[] = Array.from({ length: n }, () => new Map());
// Budget scaled by binding count — never trips where dense converges (no
// regression), still bounds the single-deeply-carried-variable adversarial
// case. undefined/0 ⇒ unlimited.
const base =
limits?.maxBlockVisits && limits.maxBlockVisits > 0 ? limits.maxBlockVisits : Infinity;
const maxUpdates = base === Infinity ? Infinity : base * nBindings;
let updates = 0;
// (block, binding) worklist, deduped via a flat pending bitmap. Visit order
// does not affect the result (each visit recomputes from current state), so
// a LIFO stack is used to avoid O(n) array shifts.
const encode = (b: number, v: number): number => b * nBindings + v;
const pending = new Uint8Array(n * nBindings);
const stack: number[] = [];
const enqueue = (b: number, v: number): void => {
const k = encode(b, v);
if (!pending[k]) {
pending[k] = 1;
stack.push(k);
}
};
// Seed: every binding genned in a block (its OUT becomes nonempty), plus the
// THROW successors of every gen block — a throwing block delivers allDefs(v)
// to its handler even when the block's own IN of v never changes (so the
// inChanged-driven throw requeue below would otherwise miss the first, static
// allDefs contribution).
for (let b = 0; b < n; b++) {
const g = gen[b];
if (!g) continue;
for (const v of g.keys()) {
enqueue(b, v);
for (const s of throwSuccs[b]) enqueue(s, v);
}
}
// Recompute IN(v) at block b from scratch, in the exact dense merge order
// (sorted predecessors; first contributor's set shared; copy-on-extend on
// subsequent contributors — never mutate a shared set). Returns the set or
// undefined when nothing reaches.
const computeIn = (b: number, v: number): DefSet | undefined => {
const p = preds[b];
if (p.length === 0) return undefined;
if (p.length === 1 && !p[0].viaThrow) return outSets[p[0].from].get(v);
let merged: DefSet | undefined;
let owned = false; // true once `merged` is our private (mutable) copy
const mergeOne = (src: DefSet | undefined): void => {
if (!src || src.size === 0) return;
if (merged === undefined) {
merged = src; // share the first contributor's set
owned = false;
return;
}
if (merged === src) return;
for (const key of src) {
if (!merged.has(key)) {
if (!owned) {
merged = new Set(merged);
owned = true;
}
merged.add(key);
}
}
};
for (const pe of p) {
if (pe.viaThrow) {
mergeOne(inSets[pe.from].get(v)); // exception may fire pre-defs…
mergeOne(allDefsGen[pe.from]?.get(v)); // …or after ANY of the block's defs
} else {
mergeOne(outSets[pe.from].get(v));
}
}
return merged;
};
// OUT(v) at b = overlay(IN): a killing gen replaces the set; a may-def-only
// gen unions without killing; no gen ⇒ OUT aliases IN.
const computeOut = (b: number, v: number, inV: DefSet | undefined): DefSet | undefined => {
const entry = gen[b]?.get(v);
if (!entry) return inV;
if (entry.kills) return entry.set;
return inV ? unionSets(inV, entry.set) : entry.set;
};
const setEq = (a: DefSet | undefined, b: DefSet | undefined): boolean => {
if (a === b) return true;
if (!a || !b || a.size !== b.size) return false;
for (const v of b) if (!a.has(v)) return false;
return true;
};
while (stack.length > 0) {
if (++updates > maxUpdates) return { converged: false };
const k = stack.pop()!;
pending[k] = 0;
const b = (k / nBindings) | 0;
const v = k - b * nBindings;
const newIn = computeIn(b, v);
const oldIn = inSets[b].get(v);
const inChanged = !setEq(oldIn, newIn);
if (newIn !== undefined) inSets[b].set(v, newIn);
const newOut = computeOut(b, v, newIn);
const oldOut = outSets[b].get(v);
const outChanged = !setEq(oldOut, newOut);
if (newOut !== undefined) outSets[b].set(v, newOut);
if (outChanged) for (const s of succs[b]) enqueue(s, v);
if (inChanged) for (const s of throwSuccs[b]) enqueue(s, v);
}
return { converged: true, inSets };
}
/**
* Statement sweep — recover statement-granular def→use facts from the per-block
* entry reaching lattices, sort them, and apply the maxFacts truncation. SHARED
@ -481,6 +663,16 @@ function sweepFacts(
selfKey !== undefined && !reaching?.has(selfKey)
? [...(reaching ?? []), selfKey]
: [...(reaching ?? [])];
// Canonical emission order (#2201 KTD6): sort each use's reaching
// def-sites by defKey (= def block, then def stmt) BEFORE the maxFacts
// cutoff. The full (untruncated) fact array is re-sorted identically at
// the end, so this is a no-op there; its purpose is to make the
// TRUNCATED subset schedule-independent — the reaching SET's insertion
// order is fixpoint-evaluation-order-dependent for loop-carried
// bindings (dense RPO vs sparse change-driven seed different keys
// first), so a pre-sort cutoff is what keeps the two solvers'
// truncated results byte-identical.
keys.sort((a, b) => a - b);
for (const key of keys) {
if (facts.length >= maxFacts) {
truncated = true;

View file

@ -24,6 +24,7 @@ import { describe, it, expect } from 'vitest';
import {
computeReachingDefs,
computeReachingDefsDense,
computeReachingDefsSparse,
type FunctionDefUse,
type ReachingDefsLimits,
} from '../../../src/core/ingestion/cfg/reaching-defs.js';
@ -393,7 +394,18 @@ interface CorpusResult {
firstFailure: string | null;
}
function runCorpus(left: Solver, right: Solver, count: number, baseSeed: number): CorpusResult {
function runCorpus(
left: Solver,
right: Solver,
count: number,
baseSeed: number,
// maxBlockVisits has DIFFERENT (intentional) semantics across the dense and
// sparse solvers — dense counts block dequeues, sparse counts (block,binding)
// dequeues — so a small budget truncates them at different points. Perturb it
// only when comparing a solver against ITSELF (same semantics); cross-solver
// byte-identity is asserted with the budget unlimited (both fully converge).
perturbBlockVisits = true,
): CorpusResult {
const flags: ShapeFlags = {
hasLoop: false,
hasThrow: false,
@ -425,7 +437,7 @@ function runCorpus(left: Solver, right: Solver, count: number, baseSeed: number)
check(cfg, undefined, `canon[${i}]`);
check(cfg, { maxFacts: 1 }, `canon[${i}]/maxFacts=1`);
check(cfg, { maxFacts: 2 }, `canon[${i}]/maxFacts=2`);
check(cfg, { maxBlockVisits: 2 }, `canon[${i}]/maxBlockVisits=2`);
if (perturbBlockVisits) check(cfg, { maxBlockVisits: 2 }, `canon[${i}]/maxBlockVisits=2`);
}
// random corpus
@ -437,7 +449,9 @@ function runCorpus(left: Solver, right: Solver, count: number, baseSeed: number)
// exercise truncation on ~1/4 of cases (small maxFacts) and the block-visit
// ceiling on ~1/8 — both must match byte-for-byte (KTD6).
if (i % 4 === 0) check(cfg, { maxFacts: 1 + (i % 3) }, `seed=${seed}/maxFacts`);
if (i % 8 === 0) check(cfg, { maxBlockVisits: 1 + (i % 4) }, `seed=${seed}/maxBlockVisits`);
if (perturbBlockVisits && i % 8 === 0) {
check(cfg, { maxBlockVisits: 1 + (i % 4) }, `seed=${seed}/maxBlockVisits`);
}
}
return { checked, flags, firstFailure };
@ -474,11 +488,58 @@ describe('#2201 reaching-defs differential equivalence', () => {
expect(a.flags).toEqual(b.flags);
});
it('PRODUCTION computeReachingDefs is byte-identical to the dense oracle', () => {
// U1: computeReachingDefs still delegates to dense, so this is trivially
// green. U5 swaps it to the sparse solver — this becomes the real gate.
const r = runCorpus(computeReachingDefs, computeReachingDefsDense, CORPUS_N, 0x5eed);
it('the SPARSE solver is byte-identical to the dense oracle (#2201 gate)', () => {
// The load-bearing equivalence gate: sparse vs dense across the full corpus,
// budget unlimited so both fully converge. maxFacts truncation IS compared
// (it must match byte-for-byte — KTD6); maxBlockVisits is not (the two count
// different things on purpose — that contrast is the no-regression test).
const r = runCorpus(
computeReachingDefsSparse,
computeReachingDefsDense,
CORPUS_N,
0x2201,
/* perturbBlockVisits */ false,
);
expect(r.firstFailure).toBeNull();
expect(r.flags.hadComputed && r.flags.hadTruncated).toBe(true);
});
it('PRODUCTION computeReachingDefs is byte-identical to the dense oracle', () => {
// U1: computeReachingDefs delegates to dense (trivially green). U5 swaps it
// to the sparse solver — this stays the production-entry gate.
const r = runCorpus(
computeReachingDefs,
computeReachingDefsDense,
CORPUS_N,
0x5eed,
/* perturbBlockVisits */ false,
);
expect(r.firstFailure).toBeNull();
});
it('sparse never regresses coverage under the production block-visit budget', () => {
// Production posture: emit passes maxBlockVisits = blocks × 64. The contract
// is one-directional — wherever the dense solver COMPUTES, the sparse solver
// must also compute and produce identical facts (no lost REACHING_DEF
// coverage). The reverse is allowed and desired: sparse may compute deep
// nests the dense solver truncates (the #2201 ceiling-stops-firing win).
let regressions = 0;
let firstRegression: string | null = null;
for (let i = 0; i < CORPUS_N; i++) {
const cfg = genCfg(0xc0de + i);
const budget = { maxBlockVisits: cfg.blocks.length * 64 };
const dense = computeReachingDefsDense(cfg, budget);
const sparse = computeReachingDefsSparse(cfg, budget);
if (dense.status === 'computed') {
const d = diffDefUse(dense, sparse);
if (d) {
regressions++;
if (!firstRegression) firstRegression = `seed=${0xc0de + i}: ${d}`;
}
}
}
expect(firstRegression).toBeNull();
expect(regressions).toBe(0);
});
});