From b53e067fc0546487f15311f653c52aeea8e7a2da Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Mon, 15 Jun 2026 18:07:09 +0000 Subject: [PATCH] bench(lbug): per-file fingerprint so the gate catches pair-file mis-routing (#2215 review) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fingerprintEmit flattened every line of every per-pair file into one array, sorted globally, and hashed — losing file boundaries, so a row routed to the WRONG pair file produced an identical fingerprint. Hash a per-file digest (filename + sha256(file bytes)) and combine the sorted entry list, so mis-routing (and within-file row reordering) now changes the fingerprint. Baseline regenerated; the new scheme yields a different hash on byte-identical emit, confirming it is sensitive to file structure the old flatten ignored. Co-Authored-By: Claude Opus 4.8 (1M context) --- gitnexus/bench/emit-persistence/baselines.json | 4 ++-- gitnexus/bench/emit-persistence/measure.mjs | 13 +++++++++---- 2 files changed, 11 insertions(+), 6 deletions(-) diff --git a/gitnexus/bench/emit-persistence/baselines.json b/gitnexus/bench/emit-persistence/baselines.json index 615d42aa8..0bb4250c3 100644 --- a/gitnexus/bench/emit-persistence/baselines.json +++ b/gitnexus/bench/emit-persistence/baselines.json @@ -1,5 +1,5 @@ { - "fingerprint": "04096714dae58fd363fa6cdbd535bcb6ba4db058b8b84d6fde504a150efaf015", + "fingerprint": "1b9dd0b783899b47067c36511d241860f291ac736e57682b0ece14148e3958ff", "scaling_budget": 1.8, - "_note": "fingerprint = order-independent sha256 of every emitted CSV line (byte-identity gate for #2203 U2/U3). scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." + "_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted — binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." } diff --git a/gitnexus/bench/emit-persistence/measure.mjs b/gitnexus/bench/emit-persistence/measure.mjs index e5ff4930b..49fbe278c 100644 --- a/gitnexus/bench/emit-persistence/measure.mjs +++ b/gitnexus/bench/emit-persistence/measure.mjs @@ -112,13 +112,18 @@ function generateGraph(entityCount) { * digest is a pure function of the emitted line SET (insertion-order agnostic). */ async function fingerprintEmit(graph, dir) { await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); - const lines = []; + // Per-file digest bound to the filename: a row routed to the WRONG pair file + // (or a header written to the wrong file) changes the fingerprint — a global + // line-flatten could not catch that. File bytes are hashed as-written (so it + // also catches within-file row reordering); the entry list is sorted so + // readdir order doesn't matter. + const entries = []; for (const name of fs.readdirSync(dir)) { if (!name.endsWith('.csv')) continue; - const text = await fsp.readFile(path.join(dir, name), 'utf8'); - for (const l of text.split('\n')) if (l.length > 0) lines.push(l); + const bytes = await fsp.readFile(path.join(dir, name)); + entries.push(`${name}\n${crypto.createHash('sha256').update(bytes).digest('hex')}`); } - return crypto.createHash('sha256').update(lines.sort().join('\n')).digest('hex'); + return crypto.createHash('sha256').update(entries.sort().join('\n')).digest('hex'); } // ---- timing ----