From e305561d34739e375af4ec207220fdf693025bda Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Mon, 15 Jun 2026 16:25:30 +0000 Subject: [PATCH] bench(lbug): emit throughput + byte-identity gate for the persistence path (#2203 U4) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Build-free bench (bench/emit-persistence/measure.mjs) times streamAllCSVsToDisk on a synthetic graph at two scales and gates: (1) an order-independent sha256 fingerprint over every emitted CSV line — the byte-identity guard for the U2/U3 emit optimisations — and (2) a scaling-ratio budget catching an O(n^2) emit re-regression. Wired into ci-tests.yml alongside the cfg/scope-capture benches. The LadybugDB COPY half needs a real DB, so its timing stays in PROF_LBUG_LOAD + the integration round-trip tests (documented in the bench README, with the deferred COPY-parallelism follow-up). Co-Authored-By: Claude Opus 4.8 (1M context) --- .github/workflows/ci-tests.yml | 9 + gitnexus/bench/emit-persistence/README.md | 56 +++++ .../bench/emit-persistence/baselines.json | 5 + gitnexus/bench/emit-persistence/measure.mjs | 202 ++++++++++++++++++ 4 files changed, 272 insertions(+) create mode 100644 gitnexus/bench/emit-persistence/README.md create mode 100644 gitnexus/bench/emit-persistence/baselines.json create mode 100644 gitnexus/bench/emit-persistence/measure.mjs diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index ee0d92332..a851696fb 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -276,6 +276,15 @@ jobs: run: node --expose-gc --import tsx bench/cfg/measure.mjs --check working-directory: gitnexus + - name: Emit-persistence throughput / byte-identity guards (#2203) + # Build-free: asserts streamAllCSVsToDisk output is byte-identical + # (order-independent CSV-line fingerprint — the #2203 U2/U3 emit + # optimisations must not change graph content) and that emit wall-time + # stays linear in node+edge count. The LadybugDB COPY half needs a real + # DB, so its timing lives in the runtime PROF_LBUG_LOAD breakdown. + run: node --import tsx bench/emit-persistence/measure.mjs --check + working-directory: gitnexus + - name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial) env: GITNEXUS_BENCH: '1' diff --git a/gitnexus/bench/emit-persistence/README.md b/gitnexus/bench/emit-persistence/README.md new file mode 100644 index 000000000..01f16de92 --- /dev/null +++ b/gitnexus/bench/emit-persistence/README.md @@ -0,0 +1,56 @@ +# Emit-persistence bench (#2203) + +Build-free throughput + byte-identity guard for the **CSV-generation half** of +the graph-DB persistence pipeline (`streamAllCSVsToDisk`), which dominates +large-repo `analyze` wall time alongside parsing (issue #2203). + +```bash +# from gitnexus/ +node --import tsx bench/emit-persistence/measure.mjs # print one JSON line +node --import tsx bench/emit-persistence/measure.mjs --check # gate vs baselines.json +``` + +## What it measures + +A synthetic `KnowledgeGraph` (files + functions + classes + 4 edge types across +the `File→Function`, `File→Class`, `Function→Function` label pairs) at two +scales: + +- **`elapsed_ms_small` / `elapsed_ms_large`** — median wall-clock over `REPS` + runs of `streamAllCSVsToDisk`. +- **`scaling_ratio`** — `(t_large/t_small)/(LARGE/SMALL)`; ~1.0 is linear. The + `--check` gate fails if it exceeds `scaling_budget` (catches an O(n²) + re-regression in the emit/routing path). +- **`fingerprint`** — order-independent sha256 over every emitted CSV line (node + CSVs + per-FROM→TO-label-pair rel CSVs). This is the **byte-identity gate**: + the U2 (direct per-pair routing) and U3 (per-row microtask elimination) + optimisations must not change graph content, and any future change that does + fails `--check`. + +## What it does NOT measure + +- **The LadybugDB `COPY` half.** Bulk loading needs a live writable DB + connection, so it can't run build-free. Its per-stage timing lives in the + runtime `PROF_LBUG_LOAD=1` breakdown (`[lbug-load prof] csv-emit=… copy-nodes=… + copy-rels=… fallback=… total=…`) and is exercised end-to-end by the + integration round-trip tests (`test/integration/basicblock-roundtrip.test.ts`, + `lbug-core-adapter.test.ts`). +- **Content extraction.** Bench nodes have no backing source files, so the + `content` column is empty — emit cost here reflects the CSV machinery + (routing, escaping, buffering, disk writes), not file reads. +- **At-scale absolute numbers.** The real postgres / kernel-`fs/` wall (issue + #2203's table) is a maintainer-run measurement; this synthetic bench is the + reproducible regression guard, not a substitute for those runs. + +## Deferred follow-up + +Parallelising the `COPY` loop (`PARALLEL=false` is load-bearing; LadybugDB is +single-writer) is **out of scope** for #2203 pending empirical validation of +concurrent-COPY support — the `PROF_LBUG_LOAD` breakdown is the prerequisite +that shows whether COPY is the dominant cost worth that risk. + +## Regenerating the baseline + +```bash +node --import tsx bench/emit-persistence/measure.mjs # copy fingerprint + ratio into baselines.json +``` diff --git a/gitnexus/bench/emit-persistence/baselines.json b/gitnexus/bench/emit-persistence/baselines.json new file mode 100644 index 000000000..615d42aa8 --- /dev/null +++ b/gitnexus/bench/emit-persistence/baselines.json @@ -0,0 +1,5 @@ +{ + "fingerprint": "04096714dae58fd363fa6cdbd535bcb6ba4db058b8b84d6fde504a150efaf015", + "scaling_budget": 1.8, + "_note": "fingerprint = order-independent sha256 of every emitted CSV line (byte-identity gate for #2203 U2/U3). scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." +} diff --git a/gitnexus/bench/emit-persistence/measure.mjs b/gitnexus/bench/emit-persistence/measure.mjs new file mode 100644 index 000000000..e5ff4930b --- /dev/null +++ b/gitnexus/bench/emit-persistence/measure.mjs @@ -0,0 +1,202 @@ +/** + * Build-free emit-path throughput + byte-identity bench for the graph-DB + * persistence pipeline (issue #2203). + * + * Measures `streamAllCSVsToDisk` — the CSV-generation half of the persistence + * path that U2 (direct per-pair relationship routing) and U3 (per-row + * microtask elimination) optimised. The LadybugDB `COPY` half needs a real DB + * connection, so its timing lives in the runtime `PROF_LBUG_LOAD` breakdown + + * the integration round-trip tests, NOT here (see README.md). + * + * For a synthetic KnowledgeGraph at two scales it reports: + * - elapsed_ms_small / elapsed_ms_large (median over REPS) + a scaling ratio + * `(t_large/t_small)/(LARGE/SMALL)`: ~1.0 linear, ~3.x quadratic; + * - an order-independent sha256 fingerprint over every emitted CSV line + * (node CSVs + per-FROM→TO-label-pair rel CSVs), as the byte-identity gate + * guarding the issue's "byte-identical graph content" requirement. + * + * Build-free: imports the `.ts` hotpaths through tsx + * (`node --import tsx bench/emit-persistence/measure.mjs`). Static `.ts` + * imports work; a top-level `await import()` breaks tsx's lexer. + * + * Without args: prints one JSON object per scenario. + * With `--check`: asserts the fingerprint == the committed baseline AND the + * scaling ratio < the recorded budget; exits non-zero on drift/regression. + */ +import fs from 'node:fs'; +import fsp from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { fileURLToPath } from 'node:url'; + +import { createKnowledgeGraph } from '../../src/core/graph/graph.ts'; +import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.ts'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +// ---- synthetic graph generation (deterministic — no randomness) ---- + +/** + * Build a graph of `entityCount` files, each with 2 functions + 1 class and 4 + * relationships. ids carry valid table-label prefixes (File:/Function:/Class:) + * so edges route to real pairs (File→Function, File→Class, Function→Function) + * — exercising the U2 router across multiple label pairs. Node `content` is + * never populated (no backing files), so emit cost reflects the CSV machinery + * (routing, escaping, buffering, disk writes), not content extraction. + */ +function generateGraph(entityCount) { + const graph = createKnowledgeGraph(); + for (let i = 0; i < entityCount; i++) { + const fp = `src/e${i}.ts`; + const fileId = `File:${fp}`; + const fnA = `Function:${fp}:fnA:1`; + const fnB = `Function:${fp}:fnB:10`; + const cls = `Class:${fp}:C:20`; + graph.addNode({ id: fileId, label: 'File', properties: { name: `e${i}.ts`, filePath: fp } }); + graph.addNode({ + id: fnA, + label: 'Function', + properties: { name: 'fnA', filePath: fp, startLine: 1, endLine: 5, isExported: true }, + }); + graph.addNode({ + id: fnB, + label: 'Function', + properties: { name: 'fnB', filePath: fp, startLine: 10, endLine: 15, isExported: false }, + }); + graph.addNode({ + id: cls, + label: 'Class', + properties: { name: 'C', filePath: fp, startLine: 20, endLine: 30, isExported: true }, + }); + graph.addRelationship({ + id: `${fileId}->${fnA}`, + sourceId: fileId, + targetId: fnA, + type: 'CONTAINS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: `${fileId}->${fnB}`, + sourceId: fileId, + targetId: fnB, + type: 'CONTAINS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: `${fileId}->${cls}`, + sourceId: fileId, + targetId: cls, + type: 'CONTAINS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: `${fnA}->${fnB}`, + sourceId: fnA, + targetId: fnB, + type: 'CALLS', + confidence: 1, + reason: '', + }); + } + return graph; +} + +// ---- byte-identity fingerprint (order-independent) ---- + +/** sha256 over every non-empty line of every emitted CSV file, sorted so the + * digest is a pure function of the emitted line SET (insertion-order agnostic). */ +async function fingerprintEmit(graph, dir) { + await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); + const lines = []; + for (const name of fs.readdirSync(dir)) { + if (!name.endsWith('.csv')) continue; + const text = await fsp.readFile(path.join(dir, name), 'utf8'); + for (const l of text.split('\n')) if (l.length > 0) lines.push(l); + } + return crypto.createHash('sha256').update(lines.sort().join('\n')).digest('hex'); +} + +// ---- timing ---- + +function median(xs) { + const s = [...xs].sort((a, b) => a - b); + const m = Math.floor(s.length / 2); + return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2; +} + +async function timeEmit(graph, dir, reps) { + await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); // warmup (not counted) + const samples = []; + for (let i = 0; i < reps; i++) { + const start = process.hrtime.bigint(); + await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); + samples.push(Number(process.hrtime.bigint() - start) / 1e6); + } + return median(samples); +} + +const SMALL = 600; +const LARGE = 2400; +const REPS = 5; + +async function measure() { + const tmpRoot = path.join(os.tmpdir(), `gitnexus-emit-bench-${process.pid}`); + await fsp.mkdir(tmpRoot, { recursive: true }); + try { + const smallGraph = generateGraph(SMALL); + const largeGraph = generateGraph(LARGE); + + const fingerprint = await fingerprintEmit(largeGraph, path.join(tmpRoot, 'fp')); + const small = await timeEmit(smallGraph, path.join(tmpRoot, 'small'), REPS); + const large = await timeEmit(largeGraph, path.join(tmpRoot, 'large'), REPS); + const scalingRatio = small > 0 ? large / small / (LARGE / SMALL) : 0; + + return { + scenario: 'streamAllCSVsToDisk', + entities_small: SMALL, + entities_large: LARGE, + nodes_large: LARGE * 4, + rels_large: LARGE * 4, + elapsed_ms_small: Number(small.toFixed(2)), + elapsed_ms_large: Number(large.toFixed(2)), + scaling_ratio: Number(scalingRatio.toFixed(3)), + fingerprint, + }; + } finally { + await fsp.rm(tmpRoot, { recursive: true, force: true }).catch(() => {}); + } +} + +// ---- run ---- + +const CHECK = process.argv.includes('--check'); +const result = await measure(); + +if (!CHECK) { + process.stdout.write(JSON.stringify(result) + '\n'); +} else { + const base = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8')); + const failures = []; + if (result.fingerprint !== base.fingerprint) { + failures.push( + `byte-identity fingerprint drift (got ${result.fingerprint}, expected ${base.fingerprint})`, + ); + } + if (result.scaling_ratio >= base.scaling_budget) { + failures.push( + `scaling ratio ${result.scaling_ratio} >= budget ${base.scaling_budget} ` + + `(${SMALL}->${LARGE} entities, ms ${result.elapsed_ms_small}->${result.elapsed_ms_large})`, + ); + } + process.stdout.write(JSON.stringify(result) + '\n'); + if (failures.length > 0) { + for (const f of failures) process.stderr.write(`[emit-persistence --check] FAIL: ${f}\n`); + process.exit(1); + } + process.stderr.write('[emit-persistence --check] PASS\n'); +}