diff --git a/gitnexus/bench/emit-persistence/baselines-streaming.json b/gitnexus/bench/emit-persistence/baselines-streaming.json index 0fabf5a02..837d78d1f 100644 --- a/gitnexus/bench/emit-persistence/baselines-streaming.json +++ b/gitnexus/bench/emit-persistence/baselines-streaming.json @@ -1,5 +1,6 @@ { - "fingerprint": "dadb6eb19a751c3d315b7c53effc6605f89e0a766b01cd5a2ae611b3720d0d55", + "fingerprint": "9408aad13feb6a65c6ce031a3264b72d1afdb1fdd1e984666fe30c3ecc24d66f", "_rebaselined_3161_static_gated_column": "Same change as baselines.json's `_rebaselined_3161_static_gated_column`: every relationship row now ends in a `staticGated` cell (`0` here, no PDG edge is a gated call), so each of the 36,000 `rel_BasicBlock_BasicBlock.csv` data rows grew by 2 bytes (`...,\"T\",0` -> `...,\"T\",0,0`) while the 7,200 BasicBlock rows are unchanged. Evidence is the inverse operation: stripping the trailing `,0` from the rel rows and re-hashing the sorted bb + rel rows gives EXACTLY the prior baseline 381de8dede253140953775c290bf9a25f82ebf3cc8ecb5250ae06a63f648b089. byte_identical_nodes / byte_identical_edges stayed true and resident_basic_blocks stayed 0 while the fingerprint gate was red, so the streamed sink still matches the whole-graph emit byte for byte and the RSS bound holds. Prior 381de8dede253140953775c290bf9a25f82ebf3cc8ecb5250ae06a63f648b089 -> dadb6eb19a751c3d315b7c53effc6605f89e0a766b01cd5a2ae611b3720d0d55.", - "_note": "Byte-identity + bounded-retention gate for streaming/chunked PDG emit (#2202). fingerprint = sha256 of the sorted, header-stripped BasicBlock + PDG-edge data rows of the canonical synthetic set. --check also asserts the streamed PdgEmitSink output is byte-identical to the whole-graph streamAllCSVsToDisk emit (byte_identical_nodes/edges) and that the in-memory graph retains 0 BasicBlocks (resident_basic_blocks === 0, the O(chunk) RSS bound). Regenerate via `node --import tsx bench/emit-persistence/measure-streaming.mjs`." + "_note": "Byte-identity + bounded-retention gate for streaming/chunked PDG emit (#2202). fingerprint = sha256 of the sorted, header-stripped BasicBlock + PDG-edge data rows of the canonical synthetic set. --check also asserts the streamed PdgEmitSink output is byte-identical to the whole-graph streamAllCSVsToDisk emit (byte_identical_nodes/edges) and that the in-memory graph retains 0 BasicBlocks (resident_basic_blocks === 0, the O(chunk) RSS bound). Regenerate via `node --import tsx bench/emit-persistence/measure-streaming.mjs`.", + "_rebaselined_3442_ladybug_null_cells": "Same NULL serialization change as baselines.json (#3442). Compared the whole-graph output with the parent writer: file set, row order and all other bytes match. The fingerprinted BasicBlock and PDG-edge files contain exactly 38,400 quoted empty cells changed to bare cells (14,400 BasicBlock cells and 24,000 edge cells); restoring their quotes reproduces the exact prior fingerprint dadb6eb19a751c3d315b7c53effc6605f89e0a766b01cd5a2ae611b3720d0d55. The new fingerprint is 9408aad13feb6a65c6ce031a3264b72d1afdb1fdd1e984666fe30c3ecc24d66f. Both streamed/whole byte-identity flags remain true and resident_basic_blocks remains 0." } diff --git a/gitnexus/bench/emit-persistence/baselines.json b/gitnexus/bench/emit-persistence/baselines.json index 9c04fb248..374ae2067 100644 --- a/gitnexus/bench/emit-persistence/baselines.json +++ b/gitnexus/bench/emit-persistence/baselines.json @@ -1,5 +1,5 @@ { - "fingerprint": "d3c3a53ead47663816344369dc0e501f31d60c462039de190ab6f5361037bd79", + "fingerprint": "2f5c46506cb8dfb79a65a276646ace766cc7645fabab8b383c89f5f0270cc30b", "_rebaselined_objective_c_node_tables": "Objective-C support adds Protocol and Category node tables to csv-generator.ts's MULTI_LANG_TYPES, so this deterministic 2,400-entity synthetic emit now creates 38 CSV files rather than 36. The two additions, category.csv and protocol.csv, are both expected 55-byte header-only files because the synthetic graph contains no Objective-C nodes. Re-emitting the graph and recomputing the fingerprint after excluding exactly those two files yields the exact prior baseline 011485e5180c005be9dfec7ac8dc4e3bb0fcfc4a680a16dcae2cf2d8af5ef7e2, proving all 36 pre-existing files retain their bytes, routing, and within-file order. Prior 011485e5180c005be9dfec7ac8dc4e3bb0fcfc4a680a16dcae2cf2d8af5ef7e2 -> d3c3a53ead47663816344369dc0e501f31d60c462039de190ab6f5361037bd79. The timing guard remains within budget: local scaling_ratio 0.965 against 1.8 and elapsed_ms_large 69.24ms against 1000ms.", "_rebaselined_3161_static_gated_column": "Every relationship row gained a trailing `staticGated` BOOLEAN column (RELATION_SCHEMA in src/core/lbug/schema.ts, REL_CSV_HEADER + buildRelRow in csv-generator.ts, the fallback CREATE in lbug-adapter.ts), written as `0` for every edge that does not carry the flag and `1` for a Zig call site inside a comptime-false branch (PR #3161). Unlike the earlier header-only rebaselines this one touches ROWS, so the evidence is the inverse operation rather than a file-set diff: re-emitting this bench's 2,400-entity graph on the branch produces the same 36 CSV files; exactly 3 of them differ from a copy with the new column stripped (`rel_File_Class.csv` 156228 -> 151416 bytes, `rel_File_Function.csv` 334008 -> 324396, `rel_Function_Function.csv` 185028 -> 180216: one header field plus `,0` per row, 4,800 / 9,600 / 4,800 rows, 2 bytes each), the other 33 files are byte-identical, and the per-file fingerprint over the stripped copies is EXACTLY the prior baseline 72096279092d4f118de7e179333705c19c9aff2664f77d71a7da48cd9f73fb5a. So no row moved between pair files and no row reordered; the only bytes that changed are the appended column. Prior 72096279092d4f118de7e179333705c19c9aff2664f77d71a7da48cd9f73fb5a -> 011485e5180c005be9dfec7ac8dc4e3bb0fcfc4a680a16dcae2cf2d8af5ef7e2. Timing gates passed while the guard was red: scaling_ratio 0.82 against the 1.8 budget, elapsed_ms_large 185.54ms against the 1000ms backstop, so no throughput claim is being rebaselined away.", "scaling_budget": 1.8, @@ -9,5 +9,6 @@ "_rebaselined_2856_property_is_detail": "Third and last of the bench guards this branch left red. The Property node table gained an `isDetail` BOOLEAN column (see PROPERTY_SCHEMA in src/core/lbug/schema.ts), so `streamAllCSVsToDisk` writes one more header field and one more cell per Property row \u2014 csv-generator.ts `propertyHeader` and the `node.label === 'Property'` tail. Verified to be header-only drift rather than a change in what is emitted: dumping every CSV this bench produces on `origin/main` and on this branch and diffing per-file (filename, byte length, sha256) shows the file SET is identical at 35 CSVs on both sides, 34 of the 35 are byte-identical, and the sole difference is `property.csv` growing 68 -> 77 bytes, `id,name,filePath,startLine,endLine,content,description,declaredType` -> `...,declaredType,isDetail`. The synthetic graph has no Property nodes, so no ROW moved at all. That is the check that matters here: a row routed to the wrong pair file, or a within-file reordering, is what this fingerprint exists to catch, and neither happened. Prior 69e9182ae205183ade24c3d8ad5d7292aea677144b1cbe443dd631bc25b0cafe -> 4ee15e742a9839671a900df4f57c1c91196c64256c8cab2ac445bec605a092d5. Both timing gates passed unchanged while this was red (scaling_ratio 0.783 vs budget 1.8, elapsed_ms_large 229ms vs the 1000ms backstop), so no throughput claim is being rebaselined away.", "_rebaselined_3040_convex_endpoint_factory": "Const and Function gained a trailing convexEndpointFactory column. A deterministic 2,400-entity emit produced the same 35 CSV files and fingerprint c4d799c5336d616955b3530ba051b7dca300d1a0e412a66741cf2f27e04c533e. Removing the new Const and Function header fields plus the new trailing empty Function cell from each of 4,800 Function rows restored the exact prior fingerprint 4ee15e742a9839671a900df4f57c1c91196c64256c8cab2ac445bec605a092d5. No file or row moved or reordered. The measured scaling ratio remained 0.826 against the 1.8 budget and elapsed_ms_large was 307.75ms against the 1000ms backstop.", "_rebaselined_3107_route_runtime_evidence": "Route gained trailing runtimeConfirmed BOOLEAN, runtimeSource STRING, and runtimeStatus STRING columns in its schema, CSV header/rows, COPY statement, and graph API projection. The deterministic emit still produces the same 35 CSV files; the synthetic benchmark graph has no Route rows, so the only byte drift is the Route CSV header and no row moved or reordered. Prior c4d799c5336d616955b3530ba051b7dca300d1a0e412a66741cf2f27e04c533e -> 7b2ec01a110dcbc66868fba2c97714aaece8c3864c2eba00df68b5c027f034d2. While the guard was red, scaling_ratio was 1.044 against the 1.8 budget and elapsed_ms_large was 121.08ms against the 1000ms backstop.", - "_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted \u2014 binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then, and record WHY in a `_rebaselined_` key alongside \u2014 bench/scope-capture/baselines.json sets that convention and it is what makes a regenerated hash reviewable. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." + "_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted \u2014 binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then, and record WHY in a `_rebaselined_` key alongside \u2014 bench/scope-capture/baselines.json sets that convention and it is what makes a regenerated hash reviewable. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`.", + "_rebaselined_3442_ladybug_null_cells": "LadybugDB 0.21.1 preserves quoted empty strings, so escapeCSVField now emits bare empty cells to retain the SQL NULL meaning of the 0.18.3 writer (#3442). Compared the canonical emit with the parent writer: all 38 file names and every row order match; exactly 31,200 quoted empty cells became bare cells in six files, with no other byte changes. Restoring quotes around those cells reproduces the exact prior fingerprint d3c3a53ead47663816344369dc0e501f31d60c462039de190ab6f5361037bd79. The new fingerprint is 2f5c46506cb8dfb79a65a276646ace766cc7645fabab8b383c89f5f0270cc30b. Timing budgets remain unchanged." } diff --git a/gitnexus/bench/incremental-write-integrity/README.md b/gitnexus/bench/incremental-write-integrity/README.md new file mode 100644 index 000000000..b5b14d881 --- /dev/null +++ b/gitnexus/bench/incremental-write-integrity/README.md @@ -0,0 +1,83 @@ +# Incremental node identity reconciliation + +From `gitnexus/`: + +```sh +node --import tsx bench/incremental-write-integrity/measure.cjs --check +node bench/incremental-write-integrity/reproduce.cjs +``` + +The first command measures the production reconciliation, including every +single-label scan and comparison of ID, name, path and source range. Construction, +CSV loading, checkpoints and correctness controls are outside the timed region. +Two warm-ups precede seven measured samples. Exact counts and SHA-256 fixture +fingerprints reject empty or incomplete work. `--check` gates normalized scaling +against the committed budget, rather than a machine-specific time limit. + +On Node v24.19.0, Linux x64, Xeon Platinum 8370C, the initial measurement was +583 ms for 16,384 nodes and 2,810 ms for 65,536 nodes per reconciliation. The +normalized scaling was 1.206 (budget 2). Analyze performs two reconciliations: +after graph COPY/checkpoint and after FTS/embedding work plus a final checkpoint, +before registration, freshness metadata or staging publication. + +The verifier covers retained nodes as well as the write set. A path/range filter +would let a corrupted field hide its own row; a count would miss swapped fields +or blank IDs. Folder lifecycle, preserved Community/Process layers and streamed +BasicBlock rows require other oracles and are excluded. Relationships and source +content are outside this identity check. + +## Native-only reduction + +`reproduce.cjs` uses only `@ladybugdb/core`, its native schema/query/COPY API, and +synthetic ASCII CSV files. There is no parser, parse cache, graph adapter, FTS, +relationship table, periodic checkpoint driver or concurrent writer. Compression +is disabled and COPY is serial, matching the production settings. + +It loads 8,192 Function rows, checkpoints, deletes 64 rows belonging to the first +and last owners, then copies back those same 64 tuples. An independent tuple-set +oracle validates the complete single-label scan before and after each operation. +It also compares affected scan rows with primary-key lookups after reopening. +Use `--keep` to retain the synthetic database and CSV files; the JSON output gives +their directory. `--require-corruption` is a diagnostic assertion, deliberately +not a CI requirement: it should fail when the native defect is fixed. + +With the original `@ladybugdb/core` 0.18.3, this reduction produced: + +| Phase | Rows | Incorrect tuples | +| ------------------------- | ----: | ---------------: | +| Initial COPY + checkpoint | 8,192 | 0 | +| DELETE | 8,128 | 0 | +| Incremental COPY | 8,192 | 3,936 | +| Explicit checkpoint | 8,192 | 3,936 | +| Read-only reopen | 8,192 | 3,936 | + +With `@ladybugdb/core` 0.21.1, the same reduction returns zero incorrect +tuples in every phase. The production reconciliation and negative controls +remain required; this result covers the reduced fixture, not a replay of the +original incident. + +For example, the scan returned an empty `id` at native offset 64, while a +primary-key lookup at that same offset returned +`Function:src/owner2.ts:fn64` with the correct name/path/range. Smaller 1,024, +2,048 and 4,096-row fixtures remained healthy in the same three-cycle adapter +probe. The affected tuple count varies with fixture size and scan shape. + +This isolates a native scan/lookup discrepancy introduced by incremental COPY +into a previously populated table. It survives checkpoint and reopen, but does +not establish physical byte loss or the precise C++ defect. The reported Unicode +change and FTS build are unnecessary for this reduction; this does not prove +that the original incident has exactly the same native cause. + +The measurement harness runs selective native writes too. An independent oracle +requires the production verifier to reject any inconsistent scan, both before +and after checkpoint/reopen; healthy scans must certify their complete identity +set. It does not quietly treat the native discrepancy as a successful graph. + +Missing-row and swapped-range negative controls separately verify that the +check fails closed. Publication tests inject damage after COPY and after FTS, +assert unchanged registry/freshness, preserve the live staged baseline, and +exercise automatic dirty-index recovery followed by a no-op run. +An in-place verification failure keeps a non-FTS dirty phase: an FTS-only +repair cannot clear the marker and recertify the inconsistent graph. +The same protection applies when FTS returns but analyzer finalization fails +before the final identity scan: recovery still rebuilds the graph. diff --git a/gitnexus/bench/incremental-write-integrity/baseline.json b/gitnexus/bench/incremental-write-integrity/baseline.json new file mode 100644 index 000000000..988196191 --- /dev/null +++ b/gitnexus/bench/incremental-write-integrity/baseline.json @@ -0,0 +1,14 @@ +{ + "version": 1, + "linear_scaling_budget": 2, + "sizes": [ + { + "nodes": 16384, + "fingerprint": "17aa2ded124af9cbd15a9fe13013410e92a478f88e63a2ad9859a05aacef8efe" + }, + { + "nodes": 65536, + "fingerprint": "a6ce82aa1d07804d16b916892dfb6ca0e6f5ce5ed606f47cb1321eb124c6ac44" + } + ] +} diff --git a/gitnexus/bench/incremental-write-integrity/measure.cjs b/gitnexus/bench/incremental-write-integrity/measure.cjs new file mode 100644 index 000000000..8db1d2d5b --- /dev/null +++ b/gitnexus/bench/incremental-write-integrity/measure.cjs @@ -0,0 +1,194 @@ +#!/usr/bin/env node +/** + * Native identity reconciliation overhead and scaling. + * node --import tsx bench/incremental-write-integrity/measure.cjs [--check] + * + * COPY, graph construction, fingerprints and correctness controls are untimed. + * Each timed sample includes all single-label scans and exact tuple checks. + * Counts/fingerprints reject empty output; timing gates use normalized scaling. + */ +const assert = require('node:assert/strict'); +const { mkdtemp, rm, readFile } = require('node:fs/promises'); +const { tmpdir, cpus } = require('node:os'); +const { join, resolve } = require('node:path'); +const { pathToFileURL } = require('node:url'); +const { createHash } = require('node:crypto'); +const ROOT = resolve(__dirname, '../..'); +const source = (file) => import(pathToFileURL(join(ROOT, file)).href); + +(async () => { + const { createKnowledgeGraph } = await source('src/core/graph/graph.ts'); + const { reconcileGraphNodeIdentities } = await source( + 'src/core/incremental/write-reconciliation.ts', + ); + const adapter = await source('src/core/lbug/lbug-adapter.ts'); + const baseline = JSON.parse(await readFile(join(__dirname, 'baseline.json'), 'utf8')); + assert.equal(baseline.version, 1); + assert.ok(Number.isFinite(baseline.linear_scaling_budget) && baseline.linear_scaling_budget > 0); + const results = []; + for (const size of baseline.sizes) { + assert.ok(Number.isSafeInteger(size.nodes) && size.nodes > 0); + assert.match(size.fingerprint, /^[a-f0-9]{64}$/); + const graph = createKnowledgeGraph(); + const hash = createHash('sha256'); + for (let i = 0; i < size.nodes; i++) { + const id = `Function:src/owner${Math.floor(i / 32)}.ts:fn${i}`; + const properties = { + name: `fn${i}`, + filePath: `src/owner${Math.floor(i / 32)}.ts`, + startLine: (i % 32) * 4, + endLine: (i % 32) * 4 + 2, + }; + graph.addNode({ id, label: 'Function', properties }); + hash.update( + JSON.stringify([ + id, + properties.name, + properties.filePath, + properties.startLine, + properties.endLine, + ]) + '\n', + ); + } + assert.equal(hash.digest('hex'), size.fingerprint); + const directory = await mkdtemp(join(tmpdir(), 'gnx-reconciliation-bench-')); + try { + await adapter.initLbug(join(directory, 'lbug'), { skipFts: true }); + await adapter.loadGraphToLbug( + graph, + directory, + directory, + undefined, + undefined, + undefined, + 'none', + ); + await adapter.tryFlushWAL(); + const samples = []; + let calls = 0; + const query = async (sql) => { + calls++; + return adapter.executeQuery(sql); + }; + for (let run = 0; run < 9; run++) { + calls = 0; + const start = performance.now(); + const receipt = await reconcileGraphNodeIdentities(graph, query, 'benchmark'); + const elapsed = performance.now() - start; + assert.equal(receipt.nodes, size.nodes); + assert.equal(calls, receipt.tables); + if (run >= 2) samples.push(elapsed); + } + const first = graph.iterNodes().next().value; + const alteredQuery = async (sql) => { + const rows = await adapter.executeQuery(sql); + if (sql.includes('(n:`Function`)')) rows.find((r) => r.id === first.id).startLine++; + return rows; + }; + await assert.rejects( + reconcileGraphNodeIdentities(graph, alteredQuery, 'negative-control'), + /startLine/, + ); + await assert.rejects( + reconcileGraphNodeIdentities(graph, async () => [], 'negative-control'), + /missing ID/, + ); + + // Native 0.18.3 can corrupt scan projections after this real write, even + // without FTS. Use an independent tuple-set oracle to require rejection + // whenever the scan is wrong; a future native fix can pass normally. + const { extractChangedSubgraph } = await source('src/core/incremental/subgraph-extract.ts'); + const files = new Set(['src/owner0.ts', `src/owner${Math.floor((size.nodes - 1) / 32)}.ts`]); + const wanted = new Set( + [...graph.iterNodes()].map((node) => + JSON.stringify([ + node.id, + node.properties.name, + node.properties.filePath, + node.properties.startLine, + node.properties.endLine, + ]), + ), + ); + const nativePhases = []; + const audit = async (phase) => { + const rows = await adapter.executeQuery( + 'MATCH (n:Function) RETURN n.id AS id, n.name AS name, ' + + 'n.filePath AS filePath, n.startLine AS startLine, n.endLine AS endLine', + ); + const actual = new Set( + rows.map((r) => JSON.stringify([r.id, r.name, r.filePath, r.startLine, r.endLine])), + ); + const mismatched = [...wanted].filter((tuple) => !actual.has(tuple)).length; + const rejected = + rows.length !== size.nodes || mismatched > 0 || actual.size !== wanted.size; + if (rejected) { + await assert.rejects( + reconcileGraphNodeIdentities(graph, adapter.executeQuery, phase), + /Graph identity reconciliation/, + ); + } else { + assert.equal( + (await reconcileGraphNodeIdentities(graph, adapter.executeQuery, phase)).nodes, + size.nodes, + ); + } + nativePhases.push({ + phase, + rows: rows.length, + missing_tuples: mismatched, + verdict: rejected ? 'rejected' : 'certified', + }); + }; + await adapter.deleteNodesForFiles([...files]); + await adapter.loadGraphToLbug( + extractChangedSubgraph(graph, files), + directory, + directory, + undefined, + undefined, + undefined, + 'none', + ); + await audit('post-COPY'); + await adapter.tryFlushWAL(); + await audit('post-checkpoint'); + await adapter.closeLbug(); + await adapter.initLbug(join(directory, 'lbug'), { readOnly: true, skipFts: true }); + await audit('reopened'); + samples.sort((a, b) => a - b); + results.push({ + nodes: size.nodes, + queries: calls, + min_ms: +samples[0].toFixed(3), + median_ms: +samples[Math.floor(samples.length / 2)].toFixed(3), + fingerprint: size.fingerprint, + native_phases: nativePhases, + }); + } finally { + await adapter.closeLbug(); + await rm(directory, { recursive: true, force: true }); + } + } + const scaling = + results[1].median_ms / results[0].median_ms / (results[1].nodes / results[0].nodes); + const report = { + node: process.version, + platform: process.platform, + cpu: cpus()[0].model, + results, + normalized_scaling: +scaling.toFixed(3), + negative_controls: 'missing/swapped fields rejected', + native_sequences: + 'selective delete/COPY, checkpoint and read-only reopen checked against independent tuple oracle', + }; + process.stdout.write(JSON.stringify(report, null, 2) + '\n'); + if (process.argv.includes('--check')) + assert.ok( + scaling <= baseline.linear_scaling_budget, + `normalized scaling ${scaling} exceeds ${baseline.linear_scaling_budget}`, + ); +})().catch((error) => { + console.error(error); + process.exitCode = 1; +}); diff --git a/gitnexus/bench/incremental-write-integrity/reproduce.cjs b/gitnexus/bench/incremental-write-integrity/reproduce.cjs new file mode 100644 index 000000000..a270d3331 --- /dev/null +++ b/gitnexus/bench/incremental-write-integrity/reproduce.cjs @@ -0,0 +1,187 @@ +#!/usr/bin/env node +/** Native-only selective COPY scan discrepancy; no parser, graph adapter or FTS. */ +const assert = require('node:assert/strict'); +const fs = require('node:fs/promises'); +const os = require('node:os'); +const path = require('node:path'); +const lbug = require('@ladybugdb/core'); + +(async () => { + const directory = await fs.mkdtemp(path.join(os.tmpdir(), 'ladybug-copy-identity-')); + let database; + let connection; + const close = async () => { + // Retire this session before closing so failures or reopening cannot close it twice. + const activeConnection = connection; + const activeDatabase = database; + connection = undefined; + database = undefined; + const errors = []; + try { + await activeConnection?.close(); + } catch (error) { + errors.push(error); + } + try { + await activeDatabase?.close(); + } catch (error) { + errors.push(error); + } + if (errors.length === 1) throw errors[0]; + if (errors.length > 1) throw new AggregateError(errors, 'Native resource cleanup failed'); + }; + const errors = []; + try { + const dbPath = path.join(directory, 'lbug'); + const count = 8192; + const tuple = (i) => [ + `Function:src/owner${Math.floor(i / 32)}.ts:fn${i}`, + `fn${i}`, + `src/owner${Math.floor(i / 32)}.ts`, + (i % 32) * 4, + (i % 32) * 4 + 2, + ]; + const indices = Array.from({ length: count }, (_, i) => i); + const expected = new Set(indices.map((i) => JSON.stringify(tuple(i)))); + const csv = (rows) => + 'id,name,filePath,startLine,endLine\n' + + rows + .map((i) => + tuple(i) + .map((value) => JSON.stringify(value)) + .join(','), + ) + .join('\n') + + '\n'; + await fs.writeFile(path.join(directory, 'full.csv'), csv(indices)); + await fs.writeFile( + path.join(directory, 'delta.csv'), + csv(indices.filter((i) => i < 32 || i >= count - 32)), + ); + + // Match GitNexus's uncompressed native constructor and serial COPY options. + const open = (readOnly) => + new lbug.Database( + dbPath, + 256 * 1024 * 1024, + false, + readOnly, + 4 * 1024 ** 3, + true, + 64 * 1024 * 1024, + true, + true, + ); + database = open(false); + connection = new lbug.Connection(database); + const query = async (cypher, params) => { + const result = params + ? await connection.execute(await connection.prepare(cypher), params) + : await connection.query(cypher); + try { + return await result.getAll(); + } finally { + await result.close(); + } + }; + const copy = (file) => + query( + `COPY Function FROM "${path.join(directory, file).replace(/\\/g, '/')}" ` + + `(HEADER=true, ESCAPE='"', DELIM=',', QUOTE='"', PARALLEL=false, auto_detect=false)`, + ); + const phases = []; + const scan = async (phase) => { + const rows = await query( + 'MATCH (n:Function) RETURN id(n) AS internalID, n.id AS id, n.name AS name, ' + + 'n.filePath AS filePath, n.startLine AS startLine, n.endLine AS endLine', + ); + const key = (r) => JSON.stringify([r.id, r.name, r.filePath, r.startLine, r.endLine]); + const bad = rows.filter((r) => !expected.has(key(r))); + const actual = new Set(rows.map(key)); + const missing = [...expected].filter((value) => !actual.has(value)).length; + phases.push({ + phase, + rows: rows.length, + wrong_tuples: bad.length, + missing_tuples: missing, + examples: bad.slice(0, 2), + }); + return { rows, bad, missing }; + }; + await query( + 'CREATE NODE TABLE Function(id STRING, name STRING, filePath STRING, ' + + 'startLine INT64, endLine INT64, PRIMARY KEY(id))', + ); + await copy('full.csv'); + await query('CHECKPOINT'); + const before = await scan('baseline'); + assert.equal(before.rows.length, count); + assert.equal(before.missing, 0); + const baselineIdsByOffset = new Map(before.rows.map((r) => [r.internalID.offset, r.id])); + await query( + "MATCH (n:Function) WHERE n.filePath IN ['src/owner0.ts', 'src/owner255.ts'] DETACH DELETE n", + ); + const deleted = await scan('after-delete'); + assert.equal(deleted.rows.length, count - 64); + assert.equal(deleted.bad.length, 0); + await copy('delta.csv'); + const copied = await scan('after-copy'); + await query('CHECKPOINT'); + await scan('after-checkpoint'); + await close(); + database = open(true); + connection = new lbug.Connection(database); + const reopened = await scan('reopened'); + const pointLookups = []; + for (const row of reopened.bad.slice(0, 2)) { + const id = baselineIdsByOffset.get(row.internalID.offset); + if (!id) continue; + pointLookups.push({ + scan: row, + lookup: await query( + 'MATCH (n:Function {id: $id}) RETURN n.id AS id, n.name AS name, n.filePath AS filePath, ' + + 'n.startLine AS startLine, n.endLine AS endLine', + { id }, + ), + }); + } + process.stdout.write( + JSON.stringify( + { + native_version: lbug.VERSION, + node: process.version, + phases, + point_lookups: pointLookups, + ...(process.argv.includes('--keep') ? { directory } : {}), + }, + null, + 2, + ) + '\n', + ); + if (process.argv.includes('--require-corruption')) { + assert.ok( + copied.bad.length > 0 || copied.missing > 0 || copied.rows.length !== count, + 'Native failure did not reproduce; this is a diagnostic, not a passing CI invariant', + ); + } + } catch (error) { + errors.push(error); + } finally { + try { + await close(); + } catch (error) { + errors.push(error); + } + try { + if (!process.argv.includes('--keep')) + await fs.rm(directory, { recursive: true, force: true }); + } catch (error) { + errors.push(error); + } + } + if (errors.length === 1) throw errors[0]; + if (errors.length > 1) throw new AggregateError(errors, 'Native benchmark failed'); +})().catch((error) => { + console.error(error); + process.exitCode = 1; +}); diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index c54323314..3b3fea7be 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -10,7 +10,7 @@ "hasInstallScript": true, "license": "PolyForm-Noncommercial-1.0.0", "dependencies": { - "@ladybugdb/core": "0.18.3", + "@ladybugdb/core": "0.21.1", "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "busboy": "^1.6.0", @@ -724,9 +724,9 @@ } }, "node_modules/@ladybugdb/core": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.3.tgz", - "integrity": "sha512-XjpPKW4MrL28D2gYGTZuIjiEcPx12L21lx58QggrdrItw8o/e9Lmg/Ejoo4Kz08lZj+rIcC1Fu9thzIYOTUlJw==", + "version": "0.21.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.21.1.tgz", + "integrity": "sha512-G0LWyn/dIqX9yHlyjpR12Rvou2o1dvCB9EkuzGa5ykCYbzur8EQhRUbi4fw3FlbckWvR6rK/tVbnFpBcRLwiHg==", "hasInstallScript": true, "license": "MIT", "dependencies": { @@ -735,17 +735,17 @@ "node-addon-api": "^6.0.0" }, "optionalDependencies": { - "@ladybugdb/core-darwin-arm64": "0.18.3", - "@ladybugdb/core-darwin-x64": "0.18.3", - "@ladybugdb/core-linux-arm64": "0.18.3", - "@ladybugdb/core-linux-x64": "0.18.3", - "@ladybugdb/core-win32-x64": "0.18.3" + "@ladybugdb/core-darwin-arm64": "0.21.1", + "@ladybugdb/core-darwin-x64": "0.21.1", + "@ladybugdb/core-linux-arm64": "0.21.1", + "@ladybugdb/core-linux-x64": "0.21.1", + "@ladybugdb/core-win32-x64": "0.21.1" } }, "node_modules/@ladybugdb/core-darwin-arm64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.3.tgz", - "integrity": "sha512-DGZTOlvSS4esEb1vTekY5IDoAvZAeYzR5cXVkECtQj9BVkk05zsvCAdTPo1Rz1BuI0qvqUVF+2WlIerI67iA2g==", + "version": "0.21.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.21.1.tgz", + "integrity": "sha512-TEFYNqBdbIojf1t29p3esgRhMDm56lD5eK91dwWFZJTVWAZoGXn+yLKs0NUQgj3lW2ZquuRTLZGglPhFEsDL1Q==", "cpu": [ "arm64" ], @@ -756,9 +756,9 @@ ] }, "node_modules/@ladybugdb/core-darwin-x64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.3.tgz", - "integrity": "sha512-Qp6j0CM/orBlK6KD0p/s4ofkIhNUwi1hdCgMw+fj81UHugWHkVLiYV4grRBdHhyplw+snchZpTxvfpxFbkG1Cw==", + "version": "0.21.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.21.1.tgz", + "integrity": "sha512-w9p3oKrqeMOeWpX3xdCc/hQJHfrZtpB9We+Ir+0qyls68dSQoeXXR1GZYzzTElOtxr4OM5USwYbVyxADx+CMZg==", "cpu": [ "x64" ], @@ -769,9 +769,9 @@ ] }, "node_modules/@ladybugdb/core-linux-arm64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.3.tgz", - "integrity": "sha512-F9miYjBuS43I7uNG199FNMqwdHJ98WA6dU3v2SZCeLXmXCdRzmYcuHQWlbNr2Tba9CX58w2XvBZoUaXZKJ/yKQ==", + "version": "0.21.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.21.1.tgz", + "integrity": "sha512-TodWmzOHxmmKvXXo9buBXGeyolxf/NqBUbo/LcR/tuMViKhGH8TBzIXi8DfVVwmsJ+CQCU065TvXgv2JoHAlZg==", "cpu": [ "arm64" ], @@ -782,9 +782,9 @@ ] }, "node_modules/@ladybugdb/core-linux-x64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.3.tgz", - "integrity": "sha512-AfG5RDp/f/IDctDMpTAT5+2MYNtlWT191xiQNjSaWD4X85DhY3Dzps8Qu5VteIAPih5d6mmoaKGs8q0XIjfkFA==", + "version": "0.21.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.21.1.tgz", + "integrity": "sha512-hCcLYv8ds0x4fkTaRkuc4f75KLdH9e0EV51ne6uBNdQSY7kyc0RgSoUG3utUGfFDMMkfgIJAeIxQY4ck9jQiXg==", "cpu": [ "x64" ], @@ -795,9 +795,9 @@ ] }, "node_modules/@ladybugdb/core-win32-x64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.3.tgz", - "integrity": "sha512-bHuFk0m9cnq0WGd9I4D8or8g6cC/BS58iatMtilqM3JpDPIQIFk6MQl6exL7P4xyWbkLwQgsrv2ToDnyoQNKvg==", + "version": "0.21.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.21.1.tgz", + "integrity": "sha512-nBS1XSRJfpexdlhkezmEiRtkUt2qEDp4zBjIhH+oxgRdD7UsRz1Jyit8foJuZAmUsoQGukN5LO+YeyDU2WZTyA==", "cpu": [ "x64" ], diff --git a/gitnexus/package.json b/gitnexus/package.json index c5ab6f3c6..1db831d0d 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -65,7 +65,7 @@ "version": "node scripts/sync-plugin-manifests.mjs" }, "dependencies": { - "@ladybugdb/core": "0.18.3", + "@ladybugdb/core": "0.21.1", "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "busboy": "^1.6.0", diff --git a/gitnexus/src/cli/detect-changes-format.ts b/gitnexus/src/cli/detect-changes-format.ts index 9e91cc330..9eece8e43 100644 --- a/gitnexus/src/cli/detect-changes-format.ts +++ b/gitnexus/src/cli/detect-changes-format.ts @@ -1,5 +1,6 @@ import { t } from './i18n/index.js'; import { formatSymbolLine } from './format-symbol.js'; +import { formatPathForTerminal } from './format-path.js'; type DetectChangesSummary = { changed_files?: number; @@ -32,6 +33,7 @@ type DetectChangesResult = { summary?: DetectChangesSummary; changed_symbols?: ChangedSymbol[]; affected_processes?: AffectedProcess[]; + unmapped_files?: string[]; }; export function formatDetectChangesResult(result: unknown): string { @@ -47,6 +49,13 @@ export function formatDetectChangesResult(result: unknown): string { // Both lead the output — a caveat printed after the summary is read too late. const notes: string[] = []; if (payload.partial) notes.push(t('tool.detectChanges.partial')); + if (payload.unmapped_files?.length) { + notes.push( + t('tool.detectChanges.unmappedSource', { + files: payload.unmapped_files.map(formatPathForTerminal).join(', '), + }), + ); + } // The plain truncation note reassures that the counts are whole. That is only // true when the run did NOT also degrade — `changed_count` sums the batches // that succeeded — so the two flags together get a different sentence. diff --git a/gitnexus/src/cli/format-path.ts b/gitnexus/src/cli/format-path.ts new file mode 100644 index 000000000..e7eab1f47 --- /dev/null +++ b/gitnexus/src/cli/format-path.ts @@ -0,0 +1,8 @@ +/** Quote control-bearing paths for terminal output; structured results retain raw paths. */ +export const formatPathForTerminal = (filePath: string): string => + /[\u0000-\u001f\u007f-\u009f]/.test(filePath) + ? JSON.stringify(filePath).replace( + /[\u007f-\u009f]/g, + (char) => `\\u${char.charCodeAt(0).toString(16).padStart(4, '0')}`, + ) + : filePath; diff --git a/gitnexus/src/cli/i18n/en.ts b/gitnexus/src/cli/i18n/en.ts index 981071f64..15114ab41 100644 --- a/gitnexus/src/cli/i18n/en.ts +++ b/gitnexus/src/cli/i18n/en.ts @@ -137,7 +137,9 @@ export const en = { 'tool.detectChanges.noOverlappingSymbols': 'Diff touched {{files}} file(s) but no indexed symbols overlap those hunks — not a clean tree.', 'tool.detectChanges.partial': - 'PARTIAL RESULT: a graph query failed, so changed symbols may be missing. Do not read this as a clean pre-commit check.', + 'PARTIAL RESULT: changed-symbol or process mapping is incomplete. Do not read this as a clean pre-commit check.', + 'tool.detectChanges.unmappedSource': + 'No symbols mapped for changed source files: {{files}}. Rebuild the index and inspect the diff; retry alone may not resolve missing or out-of-range symbols.', 'tool.detectChanges.truncated': 'LISTING CAPPED: the changed-symbol list was capped, so it does not name every changed symbol. The counts and risk level still cover all of them.', // The reassurance above is only true on its own. When the run also degraded, diff --git a/gitnexus/src/cli/i18n/zh-CN.ts b/gitnexus/src/cli/i18n/zh-CN.ts index 606a9e4e6..9519954d2 100644 --- a/gitnexus/src/cli/i18n/zh-CN.ts +++ b/gitnexus/src/cli/i18n/zh-CN.ts @@ -128,7 +128,9 @@ export const zhCN = { 'tool.detectChanges.noOverlappingSymbols': 'diff 触及 {{files}} 个文件,但没有索引符号与这些 hunk 重叠 — 并非干净工作区。', 'tool.detectChanges.partial': - '结果不完整:图查询失败,可能遗漏已变更符号。请勿将其视为通过的提交前检查。', + '结果不完整:变更符号或流程映射不完整。请勿将其视为通过的提交前检查。', + 'tool.detectChanges.unmappedSource': + '变更源码文件未映射到符号:{{files}}。请重建索引并检查 diff;仅重试可能无法解决符号缺失或源码范围不匹配。', 'tool.detectChanges.truncated': '列表已截断:已变更符号列表被截断,未列出全部变更符号。计数与风险等级仍涵盖全部符号。', 'tool.detectChanges.truncatedDegraded': diff --git a/gitnexus/src/cli/status.ts b/gitnexus/src/cli/status.ts index 26dd780b4..8316fc3ae 100644 --- a/gitnexus/src/cli/status.ts +++ b/gitnexus/src/cli/status.ts @@ -48,6 +48,7 @@ import { isFullSourceAvailable, } from '../core/content-retention.js'; import { t } from './i18n/index.js'; +import { formatPathForTerminal as formatDriftPath } from './format-path.js'; /** How many drifted paths the report names before summarizing the rest. */ const DRIFT_SAMPLE_LIMIT = 10; @@ -84,9 +85,6 @@ const describeContentDrift = (drift: IndexContentDrift | undefined) => { }; }; -/** Escape control characters in repo-relative paths before printing. */ -const formatDriftPath = (rel: string): string => - /[\u0000-\u001f\u007f]/.test(rel) ? JSON.stringify(rel) : rel; const printDriftDetail = (drift: Extract): void => { console.log( t('status.indexContentDrifted', { diff --git a/gitnexus/src/core/incremental/write-reconciliation.ts b/gitnexus/src/core/incremental/write-reconciliation.ts new file mode 100644 index 000000000..bbd717e0f --- /dev/null +++ b/gitnexus/src/core/incremental/write-reconciliation.ts @@ -0,0 +1,107 @@ +import type { GraphNode } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../graph/types.js'; +import { NODE_TABLES, type NodeTableName } from '../lbug/schema.js'; +import { sanitizeUTF8 } from '../lbug/csv-generator.js'; + +// These layers need a different oracle: folders can outlive their last file, +// derived nodes may be preserved without recomputation, and PDG can be streamed. +const EXCLUDED = new Set(['Folder', 'Community', 'Process', 'BasicBlock']); +const WITHOUT_RANGE = new Set(['File', 'Route', 'Tool']); + +interface IdentityRow { + id: string; + name: string | null; + filePath: string | null; + startLine?: number | null; + endLine?: number | null; +} + +const storedString = (value: string | undefined): string => sanitizeUTF8(value || ''); + +const storedIdentityField = (node: GraphNode, field: string): unknown => { + const value = node.properties[field]; + if (field === 'name' || field === 'filePath') { + return storedString(value as string | undefined); + } + if (node.label === 'Destination') { + return typeof value === 'number' && Number.isFinite(value) ? value : null; + } + return value ?? -1; +}; + +/** + * Certify the identity fields used by context/impact/detect_changes, including + * retained rows. A write-set-only probe misses corruption of an unchanged row. + * Scan one label at a time, without path/range predicates: a corrupt field must + * not be able to hide its own row. No multi-label scans (#3139), N+1 ID lookups, + * source-content copies, or whole-graph row materialization. + * + * This certifies node identity, not relationships or native storage internals. + * Call after COPY/checkpoint and after FTS/embedding/checkpoint, before metadata. + */ +export async function reconcileGraphNodeIdentities( + graph: KnowledgeGraph, + query: (cypher: string) => Promise, + phase: string, +): Promise<{ nodes: number; tables: number }> { + const expected = new Map>(); + for (const table of NODE_TABLES) { + if (!EXCLUDED.has(table)) expected.set(table, new Map()); + } + for (const node of graph.iterNodes()) { + const table = expected.get(node.label as NodeTableName); + if (!table) continue; + const id = storedString(node.id); + if (!id || table.has(id)) { + throw new Error(`Graph identity reconciliation (${phase}): invalid or duplicate ID ${id}`); + } + table.set(id, node); + } + + let nodes = 0; + for (const [table, remaining] of expected) { + const fields = WITHOUT_RANGE.has(table) + ? ['id', 'name', 'filePath'] + : ['id', 'name', 'filePath', 'startLine', 'endLine']; + const identityFields = fields.slice(1); + const fail: (detail: string, cause?: unknown) => never = (detail, cause) => { + throw new Error( + `Graph identity reconciliation failed (${phase}, ${table}): ${detail}. ` + + 'Freshness was not advanced; run `gitnexus analyze --force` to rebuild the graph.', + { cause }, + ); + }; + let rows: IdentityRow[]; + try { + rows = await query( + `MATCH (n:\`${table}\`) RETURN ${fields.map((f) => `n.${f} AS ${f}`).join(', ')}`, + ); + } catch (error) { + fail( + `could not read identity fields: ${error instanceof Error ? error.message : String(error)}`, + error, + ); + } + for (const row of rows) { + const node = remaining.get(row.id); + if (!node) fail(`unexpected or duplicate ID ${JSON.stringify(row.id)}`); + for (const field of identityFields) { + const wanted = storedIdentityField(node, field); + const actual = field === 'name' || field === 'filePath' ? (row[field] ?? '') : row[field]; + if (actual !== wanted) { + fail( + `${row.id}.${field}: expected ${JSON.stringify(wanted)}, read ${JSON.stringify(actual)}`, + ); + } + } + remaining.delete(row.id); + nodes++; + } + if (remaining.size) { + fail( + `${remaining.size} missing ID(s), including ${JSON.stringify(remaining.keys().next().value)}`, + ); + } + } + return { nodes, tables: expected.size }; +} diff --git a/gitnexus/src/core/ingestion/cobol-processor.ts b/gitnexus/src/core/ingestion/cobol-processor.ts index 3fac5072d..6fae16805 100644 --- a/gitnexus/src/core/ingestion/cobol-processor.ts +++ b/gitnexus/src/core/ingestion/cobol-processor.ts @@ -26,6 +26,8 @@ import { import { expandCopies } from './cobol/cobol-copy-expander.js'; import { processJclFiles } from './cobol/jcl-processor.js'; import { resolveCobolCopyTarget } from './languages/cobol/copy-target.js'; +import { COBOL_EXTENSIONS, JCL_EXTENSIONS } from './cobol/file-types.js'; +export { isCobolFile, isJclFile } from './cobol/file-types.js'; import { logger } from '../logger.js'; @@ -33,10 +35,6 @@ import { logger } from '../logger.js'; // File detection // --------------------------------------------------------------------------- -const COBOL_EXTENSIONS = new Set(['.cob', '.cbl', '.cobol', '.cpy', '.copybook']); - -const JCL_EXTENSIONS = new Set(['.jcl', '.job', '.proc']); - const COPYBOOK_EXTENSIONS = new Set(['.cpy', '.copybook']); interface CobolFile { @@ -67,16 +65,6 @@ export interface CobolProcessResult { arithmeticOps: number; } -/** Returns true if the file is a COBOL or copybook file. */ -export function isCobolFile(filePath: string): boolean { - return COBOL_EXTENSIONS.has(path.extname(filePath).toLowerCase()); -} - -/** Returns true if the file is a JCL file. */ -export function isJclFile(filePath: string): boolean { - return JCL_EXTENSIONS.has(path.extname(filePath).toLowerCase()); -} - /** Returns true if the file is a COBOL copybook. */ function isCopybook(filePath: string): boolean { return COPYBOOK_EXTENSIONS.has(path.extname(filePath).toLowerCase()); diff --git a/gitnexus/src/core/ingestion/cobol/file-types.ts b/gitnexus/src/core/ingestion/cobol/file-types.ts new file mode 100644 index 000000000..a41d622ad --- /dev/null +++ b/gitnexus/src/core/ingestion/cobol/file-types.ts @@ -0,0 +1,15 @@ +import path from 'node:path'; + +// Keep detection shared without loading the analyze-only language providers. +export const COBOL_EXTENSIONS = new Set(['.cob', '.cbl', '.cobol', '.cpy', '.copybook']); +export const JCL_EXTENSIONS = new Set(['.jcl', '.job', '.proc']); + +/** Includes COBOL programs and copybooks accepted by ingestion. */ +export function isCobolFile(filePath: string): boolean { + return COBOL_EXTENSIONS.has(path.extname(filePath).toLowerCase()); +} + +/** Includes job and procedure aliases accepted by JCL ingestion. */ +export function isJclFile(filePath: string): boolean { + return JCL_EXTENSIONS.has(path.extname(filePath).toLowerCase()); +} diff --git a/gitnexus/src/core/ingestion/pipeline-phases/routes.ts b/gitnexus/src/core/ingestion/pipeline-phases/routes.ts index 42bf639d2..cc1c48488 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/routes.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/routes.ts @@ -14,7 +14,7 @@ import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js'; import { getPhaseOutput } from './types.js'; import type { ParseOutput } from './parse.js'; -import { isBladeTemplateFilename } from 'gitnexus-shared'; +import { isTemplateRouteCandidate } from '../utils/template-file.js'; import { nextjsFileToRouteURL, normalizeFetchURL } from '../route-extractors/nextjs.js'; import { expoFileToRouteURL } from '../route-extractors/expo.js'; import { phpFileToRouteURL } from '../route-extractors/php.js'; @@ -41,6 +41,8 @@ import { readFileContents } from '../filesystem-walker.js'; import { isDev } from '../utils/env.js'; import { logger } from '../../logger.js'; +export { isTemplateRouteCandidate } from '../utils/template-file.js'; + const EXPO_NAV_PATTERNS = [ /router\.(push|replace|navigate)\(\s*['"`]([^'"`]+)['"`]/g, /]*href=\s*['"`]([^'"`]+)['"`]/g, @@ -121,17 +123,6 @@ function hasRouteParameters(routeUrl: string): boolean { return /\{[^}]+\}/.test(routeUrl); } -export const isTemplateRouteCandidate = (filePath: string): boolean => { - const normalized = filePath.replace(/\\/g, '/').toLowerCase(); - return ( - normalized.endsWith('.html') || - normalized.endsWith('.htm') || - normalized.endsWith('.ejs') || - normalized.endsWith('.hbs') || - isBladeTemplateFilename(normalized) - ); -}; - export function extractTemplateStaticFetchCalls( filePath: string, content: string, diff --git a/gitnexus/src/core/ingestion/utils/template-file.ts b/gitnexus/src/core/ingestion/utils/template-file.ts new file mode 100644 index 000000000..1024de74e --- /dev/null +++ b/gitnexus/src/core/ingestion/utils/template-file.ts @@ -0,0 +1,13 @@ +import { isBladeTemplateFilename } from 'gitnexus-shared'; + +/** Templates whose URLs and fetches can contribute route relationships to the graph. */ +export const isTemplateRouteCandidate = (filePath: string): boolean => { + const normalized = filePath.replace(/\\/g, '/').toLowerCase(); + return ( + normalized.endsWith('.html') || + normalized.endsWith('.htm') || + normalized.endsWith('.ejs') || + normalized.endsWith('.hbs') || + isBladeTemplateFilename(normalized) + ); +}; diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index aafb556f6..5230ec713 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -110,9 +110,13 @@ export const sanitizeUTF8 = (str: string): string => { }; export const escapeCSVField = (value: string | number | undefined | null): string => { - if (value === undefined || value === null) return '""'; + if (value === undefined || value === null) return ''; let str = String(value); str = sanitizeUTF8(str); + // Preserve the NULL meaning of absent strings from the 0.18.3 writer. + // Newer LadybugDB readers retain a quoted empty field as a STRING value; + // it must stay unquoted or unresolved destination addresses can join. + if (str.length === 0) return ''; return `"${str.replace(/"/g, '""')}"`; }; diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index 970832995..59f903261 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -1101,28 +1101,78 @@ export type LbugProgressCallback = (message: string) => void; /** * Run a COPY, retrying once with IGNORE_ERRORS=true (which skips row-level - * errors) on first failure. On a second failure, hand the RAW retry error to - * `onError` — each call site formats + slices its own message (#2226 F5: node - * COPY slices to 200 chars and throws; relationship COPY slices to 80 and warns, - * so the helper must not pre-format and lose that distinction). `onError` may - * throw to propagate the failure. + * errors) on first failure. Log the original failure and native COPY/warning + * receipts even when the retry succeeds. Node COPY requires every row; a + * skipped-node receipt is a failure. Call sites retain their own message + * limits and relationship fallback policy. */ const copyCsvWithRetry = async ( targetConn: lbug.Connection, copyQuery: string, onError: (retryErr: unknown) => void, + expectedRows?: number, ): Promise => { try { await queryAndDrain(targetConn, copyQuery); - } catch { + } catch (firstError) { + logger.warn( + { err: firstError, copyQuery }, + 'First COPY failure; retrying with IGNORE_ERRORS=true', + ); try { const retryQuery = copyQuery.replace( 'auto_detect=false)', 'auto_detect=false, IGNORE_ERRORS=true)', ); - await queryAndDrain(targetConn, retryQuery); + // Keep COPY and its connection-local warning receipt in one critical + // section. Only project diagnostics, never skipped_line_or_record: that + // column contains source text. Warnings are retained at a bounded native + // limit, so their count is a lower bound, not the exact skipped total. + const retry = async () => { + await drainQueryResult(await targetConn.query('CALL CLEAR_WARNINGS()')); + const result = await readQueryRows(await targetConn.query(retryQuery)); + const warnings = await readQueryRows( + await targetConn.query('CALL SHOW_WARNINGS() RETURN message, file_path, line_number'), + ); + // The native warning limit can be zero; warning rows alone cannot + // prove that every CSV row landed. Node COPY exposes its copied count + // in the result receipt even when warning retention is disabled. + const countReceipt = result + .map((row) => String(row.result ?? '')) + .map((message) => /^(\d+) tuples have been copied/.exec(message)) + .find(Boolean); + const copiedRows = countReceipt ? Number(countReceipt[1]) : undefined; + logger.warn( + { + copyQuery, + copyResult: result, + expectedRows, + copiedRows, + skippedRows: + expectedRows !== undefined && copiedRows !== undefined + ? Math.max(0, expectedRows - copiedRows) + : undefined, + retainedWarnings: warnings.length, + warningSamples: warnings.slice(0, 5), + }, + 'COPY retry completed; retained warnings describe skipped rows (a lower bound)', + ); + if (expectedRows !== undefined && (copiedRows !== expectedRows || warnings.length > 0)) { + throw new Error( + `COPY retry skipped rows or could not verify a complete node load ` + + `(copied ${copiedRows ?? 'unknown'} of ${expectedRows}; ${warnings.length} retained warning(s))`, + ); + } + }; + await (isSharedSingletonConn(targetConn) ? withConnLock(retry) : retry()); } catch (retryErr) { - onError(retryErr); + const firstMessage = firstError instanceof Error ? firstError.message : String(firstError); + const retryMessage = retryErr instanceof Error ? retryErr.message : String(retryErr); + onError( + new Error(`COPY retry failed: ${retryMessage}; first failure: ${firstMessage}`, { + cause: retryErr, + }), + ); } } }; @@ -1192,17 +1242,22 @@ const copyNodeCSVs = async ( if (!(await stagingCsvExists(csvPath))) throw missingStagingCsvError(table, csvPath, rows); const copyQuery = getCopyQuery(table, normalizeCopyPath(csvPath)); - await copyCsvWithRetry(targetConn, copyQuery, (retryErr) => { - const retryMsg = retryErr instanceof Error ? retryErr.message : String(retryErr); - // Pool exhaustion gets a remedy (#2631): the raw binder text gives the - // operator nothing to act on, and on non-4K-page hosts (Ascend aarch64, - // Apple Silicon) the pool bills up to pageSize/4KiB x faster than the - // sizing was calibrated for — name the knob and the mechanism. - const remedy = bufferPoolExhaustionRemedy(retryMsg); - throw new Error( - `COPY failed for ${table}: ${retryMsg.slice(0, 200)}${remedy ? ` ${remedy}` : ''}`, - ); - }); + await copyCsvWithRetry( + targetConn, + copyQuery, + (retryErr) => { + const retryMsg = retryErr instanceof Error ? retryErr.message : String(retryErr); + // Pool exhaustion gets a remedy (#2631): the raw binder text gives the + // operator nothing to act on, and on non-4K-page hosts (Ascend aarch64, + // Apple Silicon) the pool bills up to pageSize/4KiB x faster than the + // sizing was calibrated for — name the knob and the mechanism. + const remedy = bufferPoolExhaustionRemedy(retryMsg); + throw new Error( + `COPY failed for ${table}: ${retryMsg.slice(0, 200)}${remedy ? ` ${remedy}` : ''}`, + ); + }, + rows, + ); } }; diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 8d302c913..92925de71 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -51,6 +51,7 @@ import { import { summarizeUndecidedSatisfaction } from './ingestion/scope-resolution/undecided-satisfaction.js'; import { summarizeScopeExtractionFailures } from './ingestion/scope-resolution/scope-extraction-failures.js'; import type { KnowledgeGraph } from './graph/types.js'; +import { reconcileGraphNodeIdentities } from './incremental/write-reconciliation.js'; import { resetDegradedParseCounter } from './tree-sitter/safe-parse.js'; import { initLbug, @@ -128,6 +129,7 @@ import { resolveFtsVersionPair } from './lbug/vendored-extension-path.js'; import { startWalCheckpointDriver, checkpointOnce, + isManualCheckpointEnabled, type WalCheckpointDriver, } from './lbug/wal-checkpoint-driver.js'; import { @@ -2854,7 +2856,7 @@ async function runFullAnalysisInner( // (Bugbot review on PR #1479: a prediction that flipped post-pipeline // could skip the embedding cache load and then take the full-rebuild // path, silently losing embeddings). - const isIncremental = + const incrementalEligible = !options.force && !!existingMeta && // Belt and braces, not a second gate: the guard above already set `force` @@ -2867,6 +2869,13 @@ async function runFullAnalysisInner( repoHasGit && allFilePaths.length > 0; + // Select a full build before any selective mutation when the operator has + // disabled the checkpoint needed to certify an incremental publication. + const isIncremental = incrementalEligible && isManualCheckpointEnabled(); + if (incrementalEligible && !isIncremental) { + log('Manual WAL checkpoints are disabled; switching to a full DB write before mutation.'); + } + const hashDiff = isIncremental ? diffFileHashes(newFileHashes, existingMeta!.fileHashes) : undefined; @@ -3112,6 +3121,24 @@ async function runFullAnalysisInner( // collapse check compares the whole in-memory graph against the whole DB, // which is only a like-for-like comparison on a full rebuild. let wroteChangedSubgraphOnly = false; + const markIncrementalGraphVerification = async (): Promise => { + if (buildPath !== lbugPath) return; + const latest = await loadMeta(metaDir); + if (!latest?.incrementalInProgress) { + throw new Error('Cannot certify incremental graph without its dirty metadata marker.'); + } + // A failed or aborted identity scan is a graph failure. FTS-only repair + // and FTS crash recovery must not clear it while retaining these rows. + await saveMeta(metaDir, { + ...latest, + incrementalInProgress: { + ...latest.incrementalInProgress, + phase: 'graph-reconciliation', + updatedAt: Date.now(), + checkpointSucceeded: false, + }, + }); + }; let incrementalFtsRebuildTables: Set | undefined; if (isIncremental && hashDiff) { // ── Incremental DB writeback ─────────────────────────────────── @@ -3163,6 +3190,8 @@ async function runFullAnalysisInner( phase: string, extra: Partial> = {}, ): Promise => { + // Do not stamp live metadata while writing a staging database. + if (buildPath !== lbugPath) return; await saveMeta(metaDir, { ...existingMeta!, incrementalInProgress: { @@ -3906,6 +3935,14 @@ async function runFullAnalysisInner( 'Continuing; recovery will treat the graph-boundary checkpoint as unsuccessful.', ); } + if (wroteChangedSubgraphOnly) { + await markIncrementalGraphVerification(); + await reconcileGraphNodeIdentities( + pipelineResult.graph, + executeQuery, + 'post-COPY/checkpoint', + ); + } if (shouldStampFtsDirtyPhase(ftsWritePlan)) { // Lift the prior-meta precondition: a first-ever in-place run (Windows // full rebuild, or any in-place incremental) must stamp too. Staging @@ -4002,6 +4039,7 @@ async function runFullAnalysisInner( ? (table, indexName) => log(`FTS: ready ${table}.${indexName}`) : undefined, }); + if (wroteChangedSubgraphOnly) await markIncrementalGraphVerification(); if (ftsResult.ok) { progress('fts', 90, 'Search indexes ready'); } else if (ftsFailureIsFatal(ftsResult.failureClass, useAtomicSwap)) { @@ -4054,6 +4092,12 @@ async function runFullAnalysisInner( progress('fts', 90, 'Search indexes skipped (FTS unavailable)'); } + if (wroteChangedSubgraphOnly) { + // FTS has returned. Later embedding/finalization failures must require + // graph recovery, since post-FTS node identities are not yet certified. + await markIncrementalGraphVerification(); + } + // ── Phase 3.5: Re-insert cached embeddings ──────────────────────── // Runs on BOTH the full-rebuild path and the incremental path: // - Full rebuild / escalated write: DB was wiped, every cached row @@ -4956,96 +5000,25 @@ async function runFullAnalysisInner( // Parse-cache publish waits until after that swap + saveMeta so a failed // registerRepo / close / swap cannot replace live shards (#3153). - // Forward the --name alias and the registry-collision bypass bit. - // `allowDuplicateName` is its own concern — independent from the - // pipeline `force` above. The CLI maps it from - // `--allow-duplicate-name` only; `--force` and `--skills` both - // trigger pipeline re-run but never bypass the registry guard. - // The returned name is the one actually written to the registry - // (after applying the precedence chain in registerRepo) — reuse it - // so AGENTS.md / skill files reference the same name MCP clients - // will look up (#979). - const projectName = await registerRepo(repoPath, meta, { - name: options.registryName, - onRename: (previousName, nextName) => - log(`Registry name changed: "${previousName}" -> "${nextName}".`), - allowDuplicateName: options.allowDuplicateName, - // Non-primary branch runs upsert into the entry's branches[]; the - // primary/flat run (placement.branch === undefined) refreshes the - // top-level fields (#2106). - branch: placement.branch, - storagePath, - }); - - // ── #2354: the flat workspace slot has adopted this run's branch ────── - // Drop a now-shadowed `branches//` sub-index for the same label - // (unreachable once the flat slot serves it) and align the registry's - // top-level branch label. Best-effort (#2364 review F5): the index is - // complete and registered, and a failure - // here leaves only a stale registry label / undeleted shadowed dir — - // never wrong routing, because the flat meta this run already stamped is - // what applyBranchScope trusts. Retried by the next content-changing run - // (same-commit fast-path runs skip it: their guard compares the - // already-stamped meta label). - if (!placement.branch && branchLabel) { - try { - await adoptFlatBranchLabel(repoPath, branchLabel, storagePath); - } catch (e) { - log( - `Warning: could not sync the workspace branch label (${(e as Error).message}); continuing.`, + if (wroteChangedSubgraphOnly) { + // Registry freshness must not advance either. Include FTS, embedding + // restoration and the final WAL drain in the certified boundary. + await markIncrementalGraphVerification(); + await walCheckpointDriver.stop(); + if (!(await checkpointOnce())) { + throw new Error( + 'Graph identity reconciliation failed: final checkpoint could not be verified; run `gitnexus analyze --force`.', ); } + await reconcileGraphNodeIdentities( + pipelineResult.graph, + executeQuery, + 'pre-publish/checkpoint', + ); } - // Keep generated .gitnexus contents ignored without editing the user's root .gitignore. await ensureGitNexusIgnored(repoPath, storagePath); - // ── Generate AI context files (best-effort) ─────────────────────── - let aggregatedClusterCount = 0; - if (pipelineResult.communityResult?.communities) { - const groups = new Map(); - for (const c of pipelineResult.communityResult.communities) { - const label = c.heuristicLabel || c.label || 'Unknown'; - groups.set(label, (groups.get(label) || 0) + c.symbolCount); - } - aggregatedClusterCount = Array.from(groups.values()).filter((count) => count >= 5).length; - } - - // Only (re)generate the repo-root AI context files (AGENTS.md / CLAUDE.md / - // skills) for the primary/flat index (#2106). A non-primary branch analyze - // must not churn the repo's committed AGENTS.md with branch-specific stats. - if (!placement.branch) { - try { - await generateAIContextFiles( - repoPath, - storagePath, - projectName, - { - files: pipelineResult.totalFileCount, - nodes: stats.nodes, - edges: stats.edges, - communities: - pipelineResult.communityResult?.stats.totalCommunities ?? - existingMeta?.stats?.communities, - clusters: aggregatedClusterCount, - processes: - pipelineResult.processResult?.stats.totalProcesses ?? existingMeta?.stats?.processes, - }, - undefined, - { - skipAgentsMd: options.skipAgentsMd, - skipSkills: options.skipSkills, - noStats: options.noStats, - defaultBranch: options.defaultBranch, - hasPdg: options.pdg === true, - hasSpringActuator: options.springActuatorPath !== undefined, - }, - ); - } catch { - // Best-effort — don't fail the entire analysis for context file issues - } - } - // ── Close LadybugDB ────────────────────────────────────────────── // Stop the manual checkpoint driver before closeLbug so its // in-flight CHECKPOINT cannot race the `safeClose` CHECKPOINT. @@ -5109,6 +5082,96 @@ async function runFullAnalysisInner( // live and the next run recovers via the full-rebuild path. await saveMeta(metaDir, meta); + // Registry freshness is published only after the graph and its metadata. + // A failed close, swap, or metadata save must leave the previous registry + // receipt intact, just as it leaves the cache unpublished. + // Forward the --name alias and the registry-collision bypass bit. + // `allowDuplicateName` is its own concern — independent from the + // pipeline `force` above. The CLI maps it from + // `--allow-duplicate-name` only; `--force` and `--skills` both + // trigger pipeline re-run but never bypass the registry guard. + // The returned name is the one actually written to the registry + // (after applying the precedence chain in registerRepo) — reuse it + // so AGENTS.md / skill files reference the same name MCP clients + // will look up (#979). + const projectName = await registerRepo(repoPath, meta, { + name: options.registryName, + onRename: (previousName, nextName) => + log(`Registry name changed: "${previousName}" -> "${nextName}".`), + allowDuplicateName: options.allowDuplicateName, + // Non-primary branch runs upsert into the entry's branches[]; the + // primary/flat run (placement.branch === undefined) refreshes the + // top-level fields (#2106). + branch: placement.branch, + storagePath, + }); + + // ── #2354: the flat workspace slot has adopted this run's branch ────── + // Drop a now-shadowed `branches//` sub-index for the same label + // (unreachable once the flat slot serves it) and align the registry's + // top-level branch label. Best-effort (#2364 review F5): the index is + // complete and registered, and a failure + // here leaves only a stale registry label / undeleted shadowed dir — + // never wrong routing, because the flat meta this run already stamped is + // what applyBranchScope trusts. Retried by the next content-changing run + // (same-commit fast-path runs skip it: their guard compares the + // already-stamped meta label). + if (!placement.branch && branchLabel) { + try { + await adoptFlatBranchLabel(repoPath, branchLabel, storagePath); + } catch (e) { + log( + `Warning: could not sync the workspace branch label (${(e as Error).message}); continuing.`, + ); + } + } + + // ── Generate AI context files (best-effort) ─────────────────────── + let aggregatedClusterCount = 0; + if (pipelineResult.communityResult?.communities) { + const groups = new Map(); + for (const c of pipelineResult.communityResult.communities) { + const label = c.heuristicLabel || c.label || 'Unknown'; + groups.set(label, (groups.get(label) || 0) + c.symbolCount); + } + aggregatedClusterCount = Array.from(groups.values()).filter((count) => count >= 5).length; + } + + // Only (re)generate the repo-root AI context files (AGENTS.md / CLAUDE.md / + // skills) for the primary/flat index (#2106). A non-primary branch analyze + // must not churn the repo's committed AGENTS.md with branch-specific stats. + if (!placement.branch) { + try { + await generateAIContextFiles( + repoPath, + storagePath, + projectName, + { + files: pipelineResult.totalFileCount, + nodes: stats.nodes, + edges: stats.edges, + communities: + pipelineResult.communityResult?.stats.totalCommunities ?? + existingMeta?.stats?.communities, + clusters: aggregatedClusterCount, + processes: + pipelineResult.processResult?.stats.totalProcesses ?? existingMeta?.stats?.processes, + }, + undefined, + { + skipAgentsMd: options.skipAgentsMd, + skipSkills: options.skipSkills, + noStats: options.noStats, + defaultBranch: options.defaultBranch, + hasPdg: options.pdg === true, + hasSpringActuator: options.springActuatorPath !== undefined, + }, + ); + } catch { + // Best-effort — don't fail the entire analysis for context file issues + } + } + // Persist the incremental parse cache only after a successful graph // publish (#3153). try/catch so a cache-write failure never breaks an // otherwise successful indexing run. Prune stale chunk-hash entries first diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index 71864e674..5b26beeea 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -10,7 +10,12 @@ import { resolveGraphPath } from '../../storage/shared-store.js'; import fs from 'fs/promises'; import path from 'path'; import { createHash } from 'crypto'; -import { scoreImpactRisk, unusedAxesForImpactWalk, type ImpactRiskResult } from 'gitnexus-shared'; +import { + scoreImpactRisk, + unusedAxesForImpactWalk, + getLanguageFromFilename, + type ImpactRiskResult, +} from 'gitnexus-shared'; import { initLbug, executeQuery, @@ -30,8 +35,10 @@ import { shapeQueryProcessAttaches } from './query-process-attaches.js'; import { LBUG_ID_PROBE_BATCH_SIZE, LBUG_QUERY_BATCH_SIZE } from '../../core/lbug/query-batch.js'; import { chunk, mapConcurrent } from '../../lib/utils.js'; import { pathSuffixOf } from './path-predicate.js'; +import { isCobolFile, isJclFile } from '../../core/ingestion/cobol/file-types.js'; import { toOneBasedLine } from '../../core/ingestion/utils/line-base.js'; import { isTestFilePath } from '../../core/ingestion/utils/test-file-path.js'; +import { isTemplateRouteCandidate } from '../../core/ingestion/utils/template-file.js'; import { isWalCorruptionError, WAL_RECOVERY_SUGGESTION } from '../../core/lbug/lbug-config.js'; // Embedding imports are lazy (dynamic import) to avoid loading onnxruntime-node // at MCP server startup — crashes on unsupported Node ABI versions (#89) @@ -6445,6 +6452,7 @@ export class LocalBackend { // throwaway arrays the size of the row set (40k rows 11.4ms → 4.5ms, 200k // rows 71.3ms → 26.6ms). const exactlyMatchedPaths = new Set(); + const mappedPaths = new Set(); for (const row of symbolRows) { if (row.filePath === row.diffPath) exactlyMatchedPaths.add(row.diffPath); } @@ -6454,6 +6462,8 @@ export class LocalBackend { if (sym.filePath !== sym.diffPath && exactlyMatchedPaths.has(diffPath)) continue; const hunks = hunksByPath.get(diffPath) ?? []; if (!hunksOverlapRange(hunks, sym.startLine, sym.endLine)) continue; + // A suffix fallback is a hint, not proof that this is the changed file. + if (sym.filePath === diffPath) mappedPaths.add(diffPath); if (changedSymbols.has(sym.id)) continue; changedSymbols.set(sym.id, { @@ -6465,6 +6475,27 @@ export class LocalBackend { }); } + // An empty successful query cannot prove a source diff is safe: its rows + // may be missing, outside indexed spans, or not yet indexed. Keep ordinary + // docs/config diffs measurable, but withhold a ranked source-risk verdict. + const isSourceFile = (file: string): boolean => + getLanguageFromFilename(file) !== null || + isCobolFile(file) || + isJclFile(file) || + isTemplateRouteCandidate(file); + const unmappedFiles = [ + ...new Set( + fileDiffs + .filter( + ({ filePath, oldFilePath }) => + !mappedPaths.has(filePath) && + (isSourceFile(filePath) || (oldFilePath !== undefined && isSourceFile(oldFilePath))), + ) + .map(({ filePath }) => filePath), + ), + ]; + if (unmappedFiles.length > 0) queryDegraded = true; + // Find affected processes -- batched queries instead of N+1 const affectedProcesses = new Map(); if (changedSymbols.size > 0) { @@ -6565,8 +6596,9 @@ export class LocalBackend { }, changed_symbols: listedSymbols, affected_processes: Array.from(affectedProcesses.values()), - // A swallowed query failure makes the counts/risk above incomplete — tell - // the caller so the safety gate isn't trusted as a clean result (#2283). + ...(unmappedFiles.length > 0 && { unmapped_files: unmappedFiles }), + // Failed queries or unmapped source files leave counts/risk incomplete; + // the safety gate must not treat that as a clean result (#2283). ...(queryDegraded && { partial: true }), ...(listedSymbols.length < changedSymbols.size && { truncated: true }), }; diff --git a/gitnexus/src/mcp/tools.ts b/gitnexus/src/mcp/tools.ts index 2965fcd6f..77b527e00 100644 --- a/gitnexus/src/mcp/tools.ts +++ b/gitnexus/src/mcp/tools.ts @@ -400,7 +400,7 @@ AFTER THIS: Review affected processes. Use context() on high-risk symbols. READ GIT WORKTREE SUPPORT: GitNexus automatically detects when the MCP server was launched from inside a linked git worktree and runs git diff against that worktree — no extra parameters needed in the common case. Pass "worktree" explicitly only when the server was started from a different directory than the worktree you are editing (e.g., the server runs from the canonical root but your changes are in a linked worktree at a different path). Returns: changed symbols, affected processes, and a risk summary. -- partial: true — a step failed and was swallowed, so the result is incomplete and risk_level is "unknown" instead of a ranked level. Two causes, with different blast radii: the symbol query (or an unparseable diff) degrades everything — changed_symbols, both counts, and the processes derived from them — while a failed process lookup degrades only affected_processes and the risk read off it, leaving the changed-symbol counts sound. changed_count:0 with partial:true is NOT a clean pre-commit check; re-run before treating the diff as safe. +- partial: true — mapping is incomplete, so risk_level is "unknown" instead of a ranked level. A failed symbol query (or an unparseable diff) degrades changed_symbols, both counts, and derived processes; a failed process lookup degrades only affected_processes and risk, leaving changed-symbol counts sound. unmapped_files lists changed supported source files with no mapped symbols, including source renames without hunks: their symbols may be missing, unindexed, or outside indexed ranges even when every query succeeded. Rebuild the index and inspect those diffs; retry alone may not resolve this state. changed_count:0 with partial:true is NOT a clean pre-commit check. - truncated: true — the changed_symbols LISTING was capped for this response. summary.changed_count counts every symbol the run observed: the true total normally, a LOWER BOUND when partial:true. Compare it with the array length rather than trusting the array.`, annotations: READ_ONLY_TOOL_ANNOTATIONS, inputSchema: { diff --git a/gitnexus/src/server/api.ts b/gitnexus/src/server/api.ts index 0d554125a..86cafb0c1 100644 --- a/gitnexus/src/server/api.ts +++ b/gitnexus/src/server/api.ts @@ -550,7 +550,7 @@ const mapGraphRelationshipRow = (row: any): GraphRelationship => ({ sourceId: row.sourceId, targetId: row.targetId, confidence: row.confidence, - reason: row.reason, + reason: row.reason ?? '', step: row.step, }); diff --git a/gitnexus/src/storage/git.ts b/gitnexus/src/storage/git.ts index 0d0e138e4..f4e26b99c 100644 --- a/gitnexus/src/storage/git.ts +++ b/gitnexus/src/storage/git.ts @@ -827,6 +827,8 @@ export interface DiffHunk { export interface FileDiff { filePath: string; + /** Decoded pre-rename path; either side can identify a source-file change. */ + oldFilePath?: string; hunks: DiffHunk[]; } @@ -1061,7 +1063,10 @@ export function parseDiffHunksResult(diffOutput: string): DiffHunkParseResult { } else { unparsedGitHeaders++; } - } else if (line.startsWith('rename to ')) { + } else if (!inHunk && line.startsWith('rename from ')) { + const oldFilePath = decodeGitPathToken(line.slice('rename from '.length)); + if (current && oldFilePath) current.oldFilePath = oldFilePath; + } else if (!inHunk && line.startsWith('rename to ')) { const filePath = pathFromRenameTo(line); if (!filePath) continue; if (current) current.filePath = filePath; diff --git a/gitnexus/test/integration/analyze-atomic-swap.test.ts b/gitnexus/test/integration/analyze-atomic-swap.test.ts index f664a6bb6..9731bf51b 100644 --- a/gitnexus/test/integration/analyze-atomic-swap.test.ts +++ b/gitnexus/test/integration/analyze-atomic-swap.test.ts @@ -19,32 +19,55 @@ import { promises as fs } from 'node:fs'; import path from 'node:path'; type LbugAdapter = typeof import('../../src/core/lbug/lbug-adapter.js'); +type RepoManager = typeof import('../../src/storage/repo-manager.js'); +type FsAtomic = typeof import('../../src/storage/fs-atomic.js'); const ctx = vi.hoisted(() => ({ loadMock: vi.fn(), realLoad: null as LbugAdapter['loadGraphToLbug'] | null, deleteMock: vi.fn(), realDelete: null as LbugAdapter['deleteNodesForFiles'] | null, + closeMock: vi.fn(), + realClose: null as LbugAdapter['closeLbug'] | null, + saveMetaMock: vi.fn(), + realSaveMeta: null as RepoManager['saveMeta'] | null, + renameMock: vi.fn(), + realRename: null as FsAtomic['retryRename'] | null, })); -// Delegating mock: overrides only loadGraphToLbug so a rebuild can be made to -// fail on demand (mirrors run-analyze-adopt-failure.test.ts). +// Delegating mocks keep real graph, metadata, and registry writes while +// allowing failures at the write and publication boundaries. vi.mock('../../src/core/lbug/lbug-adapter.js', async (importOriginal) => { const actual = await importOriginal(); ctx.realLoad = actual.loadGraphToLbug; ctx.realDelete = actual.deleteNodesForFiles; + ctx.realClose = actual.closeLbug; ctx.loadMock.mockImplementation(actual.loadGraphToLbug); ctx.deleteMock.mockImplementation(actual.deleteNodesForFiles); + ctx.closeMock.mockImplementation(actual.closeLbug); return { ...actual, loadGraphToLbug: ctx.loadMock, deleteNodesForFiles: ctx.deleteMock, + closeLbug: ctx.closeMock, }; }); +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + ctx.realSaveMeta = actual.saveMeta; + ctx.saveMetaMock.mockImplementation(actual.saveMeta); + return { ...actual, saveMeta: ctx.saveMetaMock }; +}); +vi.mock('../../src/storage/fs-atomic.js', async (importOriginal) => { + const actual = await importOriginal(); + ctx.realRename = actual.retryRename; + ctx.renameMock.mockImplementation(actual.retryRename); + return { ...actual, retryRename: ctx.renameMock }; +}); import { analyzeFailureMayHaveMutatedLiveIndex, runFullAnalysis, } from '../../src/core/run-analyze.js'; -import { getStoragePaths } from '../../src/storage/repo-manager.js'; +import { getStoragePaths, loadMeta, readRegistry } from '../../src/storage/repo-manager.js'; import { initLbug as poolInit, executeQuery as poolQuery, @@ -83,6 +106,16 @@ describe.skipIf(isWin)('atomic full-rebuild swap (#2)', () => { ctx.deleteMock.mockImplementation((...a: Parameters) => ctx.realDelete!(...a), ); + ctx.closeMock.mockReset(); + ctx.closeMock.mockImplementation(() => ctx.realClose!()); + ctx.saveMetaMock.mockReset(); + ctx.saveMetaMock.mockImplementation((...a: Parameters) => + ctx.realSaveMeta!(...a), + ); + ctx.renameMock.mockReset(); + ctx.renameMock.mockImplementation((...a: Parameters) => + ctx.realRename!(...a), + ); }); afterEach(async () => { @@ -144,6 +177,102 @@ describe.skipIf(isWin)('atomic full-rebuild swap (#2)', () => { } }, 180_000); + it.each( + (['full', 'incremental'] as const).flatMap((writeMode) => + (['close', 'swap', 'metadata'] as const).map((failurePoint) => ({ + writeMode, + failurePoint, + })), + ), + )( + 'keeps registry freshness unchanged when $writeMode publication fails at $failurePoint', + async ({ writeMode, failurePoint }) => { + const { repo, cleanup } = await makeRepo(); + const repoId = `atomic-publication-${writeMode}-${failurePoint}`; + try { + await runFullAnalysis(repo, {}, { onProgress: () => {} }); + const { lbugPath, storagePath } = getStoragePaths(repo); + const oldGraph = await fs.readFile(lbugPath); + const oldMeta = await loadMeta(storagePath); + const oldRegistry = await readRegistry(); + expect(oldRegistry).toHaveLength(1); + expect(oldRegistry[0]).toMatchObject({ + lastCommit: oldMeta!.lastCommit, + indexedAt: oldMeta!.indexedAt, + }); + + await fs.writeFile( + path.join(repo, 'a.ts'), + 'export function replacement() { return "new graph"; }\n', + ); + execSync('git -c user.name=t -c user.email=t@t commit -am replacement', { + cwd: repo, + stdio: 'pipe', + }); + const nextCommit = execSync('git rev-parse HEAD', { + cwd: repo, + encoding: 'utf8', + }).trim(); + expect(nextCommit).not.toBe(oldMeta!.lastCommit); + + const injected = new Error(`injected final ${failurePoint} failure`); + ctx.loadMock.mockClear(); + ctx.deleteMock.mockClear(); + if (failurePoint === 'close') { + ctx.closeMock.mockImplementation(async () => { + // Actually release native handles, then inject the rejection only + // after loading the new graph, not at the pre-rebuild close. + await ctx.realClose!(); + if (ctx.loadMock.mock.calls.length > 0) throw injected; + }); + } else if (failurePoint === 'swap') { + ctx.renameMock.mockImplementation( + async (...args: Parameters) => { + const [from, to] = args; + if (from.startsWith(`${lbugPath}.staging.`) && to === lbugPath) throw injected; + await ctx.realRename!(...args); + }, + ); + } else { + ctx.saveMetaMock.mockImplementation( + async (...args: Parameters) => { + const [, meta] = args; + if (meta.lastCommit === nextCommit && !meta.incrementalInProgress) throw injected; + await ctx.realSaveMeta!(...args); + }, + ); + } + + const failure = await runFullAnalysis( + repo, + writeMode === 'full' ? { force: true } : { atomicIncremental: true }, + { onProgress: () => {} }, + ).catch((error: unknown) => error); + expect(failure).toBe(injected); + expect(ctx.loadMock).toHaveBeenCalled(); + expect(ctx.deleteMock.mock.calls.length > 0).toBe(writeMode === 'incremental'); + expect(await readRegistry()).toEqual(oldRegistry); + expect(await loadMeta(storagePath)).toMatchObject({ lastCommit: oldMeta!.lastCommit }); + expect(await lingeringTemp(lbugPath)).toEqual([]); + + // A metadata failure occurs after the swap; close/swap failures keep + // the old bytes. Both cases must retain the old registry receipt. + const published = failurePoint === 'metadata'; + expect(analyzeFailureMayHaveMutatedLiveIndex(failure)).toBe(published); + expect((await fs.readFile(lbugPath)).equals(oldGraph)).toBe(!published); + await poolInit(repoId, lbugPath); + const names = (await poolQuery(repoId, 'MATCH (f:Function) RETURN f.name AS n')).flatMap( + (row) => Object.values(row as Record).map(String), + ); + expect(names.sort()).toEqual(published ? ['replacement'] : ['caller', 'greet']); + } finally { + await poolClose(repoId); + await cleanup(); + } + }, + 180_000, + ); + it('marks a failure after an atomic publish as potentially live-mutating', async () => { const { repo, cleanup } = await makeRepo(); try { diff --git a/gitnexus/test/integration/detect-changes-path-anchoring.test.ts b/gitnexus/test/integration/detect-changes-path-anchoring.test.ts index f4bd381d8..f65cdec6b 100644 --- a/gitnexus/test/integration/detect-changes-path-anchoring.test.ts +++ b/gitnexus/test/integration/detect-changes-path-anchoring.test.ts @@ -14,6 +14,7 @@ * 0-based line space (#2377, #2915). */ import { it, expect, beforeAll, vi } from 'vitest'; +import { execFileSync } from 'node:child_process'; import { mkdirSync, writeFileSync } from 'fs'; import path from 'path'; import { LocalBackend } from '../../src/mcp/local/local-backend.js'; @@ -58,8 +59,10 @@ function makeWorkingCopy(): string { /** The fields these tests read off one `detect_changes` run. */ type DetectChangesResult = { error?: unknown; - summary: { changed_count: number }; + summary: { changed_count: number; risk_level: string }; changed_symbols: { name: string; filePath: string }[]; + partial?: boolean; + unmapped_files?: string[]; }; withTestLbugDB( @@ -118,3 +121,51 @@ withTestLbugDB( }, }, ); + +withTestLbugDB( + 'detect-changes-unindexed-sibling', + (handle) => { + it('does not certify a new root index.ts from an indexed pkg/index.ts suffix match', async () => { + const backend = (handle as typeof handle & { _backend: LocalBackend })._backend; + const result = (await backend.callTool('detect_changes', { + scope: 'staged', + })) as DetectChangesResult; + expect(result.error).toBeUndefined(); + // The fallback remains visible, but cannot certify the identity of the new file. + expect(result.changed_symbols).toEqual([ + expect.objectContaining({ name: 'helper', filePath: 'pkg/index.ts' }), + ]); + expect(result.summary.risk_level).toBe('unknown'); + expect(result.partial).toBe(true); + expect(result.unmapped_files).toEqual(['index.ts']); + }); + }, + { + seed: [ + "CREATE (fn:Function {id: 'Function:pkg/index.ts:helper', name: 'helper', filePath: 'pkg/index.ts', startLine: 0, endLine: 1, isExported: true})", + ], + poolAdapter: true, + afterSetup: async (handle) => { + const repoDir = tempDirs.dir(); + mkdirSync(path.join(repoDir, 'pkg'), { recursive: true }); + writeFileSync(path.join(repoDir, 'pkg/index.ts'), 'export function helper() {\n}\n'); + initGitRepo(repoDir); + commitAll(repoDir, 'indexed sibling'); + writeFileSync(path.join(repoDir, 'index.ts'), 'export function added() {\n}\n'); + execFileSync('git', ['add', 'index.ts'], { cwd: repoDir }); + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'sibling-repo', + path: repoDir, + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'abc1234', + stats: { files: 1, nodes: 1, edges: 0, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as typeof handle & { _backend: LocalBackend })._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/fts-extension-e2e.test.ts b/gitnexus/test/integration/fts-extension-e2e.test.ts index 47f79d137..9bc6ad544 100644 --- a/gitnexus/test/integration/fts-extension-e2e.test.ts +++ b/gitnexus/test/integration/fts-extension-e2e.test.ts @@ -25,6 +25,7 @@ import { spawnSync } from 'child_process'; import path from 'path'; import fs from 'fs'; import os from 'os'; +import { inspect } from 'node:util'; import { getExtensionInstallChildProcessArgs } from '../../src/core/lbug/extension-loader.js'; import { @@ -144,6 +145,8 @@ const makeFixtureRepo = (label: string): string => { interface CliResult { status: number | null; + /** Keep native termination and spawn failures visible in assertion messages. */ + diagnostics: string; /** stdout + stderr combined — warn lines and progress renderer interleave streams. */ output: string; } @@ -172,7 +175,20 @@ const runCli = ( NODE_OPTIONS: `${process.env.NODE_OPTIONS || ''} --max-old-space-size=8192`.trim(), }, }); - return { status: result.status, output: `${result.stdout ?? ''}\n${result.stderr ?? ''}` }; + return { + status: result.status, + diagnostics: inspect( + { + status: result.status, + signal: result.signal, + error: result.error, + stdout: result.stdout, + stderr: result.stderr, + }, + { depth: null, maxStringLength: null }, + ), + output: `${result.stdout ?? ''}\n${result.stderr ?? ''}`, + }; }; beforeAll(() => { @@ -208,7 +224,7 @@ describe('happy path — extension pre-installed, fully offline (load-only)', () it('analyze builds the index with FTS and emits no degradation warning', () => { const result = runCli(['analyze'], repo, home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('indexed successfully'); expect(result.output).not.toContain('FTS extension unavailable'); expect(result.output).not.toContain('search is disabled'); @@ -216,14 +232,14 @@ describe('happy path — extension pre-installed, fully offline (load-only)', () it('query finds the symbol via BM25 with no degradation warning', () => { const result = runCli(['query', 'greetE2eSymbol'], repo, home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('greetE2eSymbol'); expect(result.output).not.toContain('keyword search degraded'); }, 60_000); it('doctor reports a live-probed available FTS and a resolved LadybugDB version', () => { const result = runCli(['doctor'], repo, home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('Full-text search: available'); // #2374: version used to print as "unknown" on every platform. expect(result.output).toMatch(/LadybugDB:\s*\d+\.\d+\.\d+/); @@ -231,7 +247,7 @@ describe('happy path — extension pre-installed, fully offline (load-only)', () it('analyze --repair-fts rebuilds the search indexes offline', () => { const result = runCli(['analyze', '--repair-fts'], repo, home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('FTS indexes repaired successfully'); }, 180_000); }); @@ -250,7 +266,7 @@ describe('packaged vendor survives a broken or missing home copy', () => { it('analyze stays FTS-available when ~/.lbdb is broken', () => { const result = runCli(['analyze'], repo, home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('indexed successfully'); expect(result.output).not.toContain('FTS extension unavailable'); expect(result.output).not.toContain('search is disabled'); @@ -258,13 +274,13 @@ describe('packaged vendor survives a broken or missing home copy', () => { it('analyze --repair-fts succeeds from the packaged artifact', () => { const result = runCli(['analyze', '--repair-fts'], repo, home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('FTS indexes repaired successfully'); }, 180_000); it('query finds the symbol with no HOME-copy degradation warning', () => { const result = runCli(['query', 'greetE2eSymbol'], repo, home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('greetE2eSymbol'); expect(result.output).not.toContain('keyword search degraded'); expect(result.output).not.toContain('FTS extension failed to load'); @@ -272,7 +288,7 @@ describe('packaged vendor survives a broken or missing home copy', () => { it('doctor reports a live-probed available FTS despite a broken HOME copy', () => { const result = runCli(['doctor'], repo, home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('Full-text search: available'); }, 60_000); @@ -280,7 +296,7 @@ describe('packaged vendor survives a broken or missing home copy', () => { const missing = makeHome('missing'); const missingRepo = makeFixtureRepo('missing'); const result = runCli(['analyze'], missingRepo, missing.home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('indexed successfully'); expect(result.output).not.toContain('FTS extension unavailable'); expect(result.output).not.toContain('has not been installed'); @@ -295,7 +311,7 @@ describe('regression — the home copy disappears between analyze runs (#2841)', // 1. First analyze with the extension in place: the index ends up carrying // an FTS index on every searchable table. const first = runCli(['analyze'], repo, home, 'load-only'); - expect(first.status).toBe(0); + expect(first.status, first.diagnostics).toBe(0); // This case needs run 1 to actually BUILD the indexes — without them there // is nothing for the gate to trip on and the assertions below would be // vacuous. When the seeded extension cannot load on this host (the same @@ -332,7 +348,7 @@ describe('regression — the home copy disappears between analyze runs (#2841)', // table File but its extension is not loaded" and no mention of FTS at all. // Packaged vendor still loads after HOME vanishes, so incremental stays // incremental (no Binder, no full-DB escalation). - expect(second.status).toBe(0); + expect(second.status, second.diagnostics).toBe(0); expect(second.output).not.toContain('its extension is not loaded'); expect(second.output).not.toContain('full DB write'); expect(second.output).not.toContain('forcing full rebuild'); @@ -346,15 +362,15 @@ describe('auto policy — packaged vendor does not need a HOME reinstall', () => const repo = makeFixtureRepo('heal'); const first = runCli(['analyze'], repo, home, 'load-only'); - expect(first.status).toBe(0); + expect(first.status, first.diagnostics).toBe(0); expect(first.output).not.toContain('FTS extension unavailable'); const repair = runCli(['analyze', '--repair-fts'], repo, home, 'auto'); - expect(repair.status).toBe(0); + expect(repair.status, repair.diagnostics).toBe(0); expect(repair.output).toContain('FTS indexes repaired successfully'); const query = runCli(['query', 'greetE2eSymbol'], repo, home, 'load-only'); - expect(query.status).toBe(0); + expect(query.status, query.diagnostics).toBe(0); expect(query.output).toContain('greetE2eSymbol'); expect(query.output).not.toContain('keyword search degraded'); }, 600_000); @@ -363,7 +379,7 @@ describe('auto policy — packaged vendor does not need a HOME reinstall', () => const { home } = makeHome('missing'); const repo = makeFixtureRepo('fresh'); const result = runCli(['analyze'], repo, home, 'load-only'); - expect(result.status).toBe(0); + expect(result.status, result.diagnostics).toBe(0); expect(result.output).toContain('indexed successfully'); expect(result.output).not.toContain('FTS extension unavailable'); }, 600_000); diff --git a/gitnexus/test/integration/lbug-core-adapter.test.ts b/gitnexus/test/integration/lbug-core-adapter.test.ts index 814fd60a4..89dd1c8a8 100644 --- a/gitnexus/test/integration/lbug-core-adapter.test.ts +++ b/gitnexus/test/integration/lbug-core-adapter.test.ts @@ -11,6 +11,8 @@ import { describe, it, expect } from 'vitest'; import fs from 'fs/promises'; import path from 'path'; +import { PassThrough } from 'node:stream'; +import type { Response } from 'express'; import type { GraphRelationship } from 'gitnexus-shared'; import { withTestLbugDB } from '../helpers/test-indexed-db.js'; import { skipUnlessFtsAvailable } from '../helpers/fts-availability.js'; @@ -61,6 +63,39 @@ withTestLbugDB( expect(folderRows).toHaveLength(1); }); + it('serializes CSV-loaded empty relationship reasons as strings in both graph APIs', async () => { + const { executeQuery } = await import('../../src/core/lbug/lbug-adapter.js'); + const { buildGraph, streamGraphNdjson } = await import('../../src/server/api.js'); + const stored = await executeQuery( + 'MATCH ()-[r:CodeRelation]->() RETURN r.reason AS reason', + ); + expect(stored).toEqual(Array.from({ length: 4 }, () => ({ reason: null }))); + + const buffered = JSON.parse(JSON.stringify(await buildGraph())); + const response = new PassThrough(); + let ndjson = ''; + response.on('data', (chunk: Buffer) => { + ndjson += chunk.toString(); + }); + try { + await streamGraphNdjson(response as unknown as Response); + } finally { + response.destroy(); + } + const streamed = ndjson + .trim() + .split('\n') + .map((line) => JSON.parse(line)) + .filter((record) => record.type === 'relationship') + .map((record) => record.data); + expect(buffered.relationships).toHaveLength(4); + expect({ + buffered: buffered.relationships.map((rel: GraphRelationship) => rel.reason), + streamed: streamed.map((rel: GraphRelationship) => rel.reason), + }).toEqual({ buffered: ['', '', '', ''], streamed: ['', '', '', ''] }); + expect(streamed).toEqual(expect.arrayContaining(buffered.relationships)); + }); + it('createFTSIndex: creates FTS index on Function table without error', async (ctx) => { await skipUnlessFtsAvailable(ctx); const { createFTSIndex } = await import('../../src/core/lbug/lbug-adapter.js'); diff --git a/gitnexus/test/integration/lbug-interrupted-checkpoint-recovery.test.ts b/gitnexus/test/integration/lbug-interrupted-checkpoint-recovery.test.ts index d2dea28cf..287f2333b 100644 --- a/gitnexus/test/integration/lbug-interrupted-checkpoint-recovery.test.ts +++ b/gitnexus/test/integration/lbug-interrupted-checkpoint-recovery.test.ts @@ -20,58 +20,28 @@ * and the behavioral contract is held on every pin by the mocked * forced-refusal suites instead. */ -import { afterAll, describe, expect, it } from 'vitest'; +import { afterAll, describe, expect, it, vi } from 'vitest'; +import { spawnSync } from 'node:child_process'; import fs from 'node:fs/promises'; import os from 'node:os'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; import { closeLbug, executeQuery, initLbug } from '../../src/core/lbug/pool-adapter.js'; import lbug from '@ladybugdb/core'; +import { closeQueryResults } from '../../src/core/lbug/query-result-utils.js'; const REPO = 'test-interrupted-checkpoint'; const ROWS = 300; /** - * Windows: the native close() resolves before the kernel releases the file's - * handles and byte-range locks — the next open then dies with Win32 Error 33 - * ("another process has locked a portion of the file"), which is exactly how - * this fixture failed its first hosted run. Probe-read both the db and its - * residual WAL until the engine's locks are gone. Bounded, so a real handle - * leak fails loudly instead of hanging; a pass-through on POSIX (first probe - * always succeeds). - */ -async function waitForFixtureRelease(dbPath: string): Promise { - for (const target of [dbPath, `${dbPath}.wal`]) { - for (let attempt = 0; ; attempt++) { - try { - const fh = await fs.open(target, 'r'); - try { - await fh.read(Buffer.alloc(1), 0, 1, 0); - } finally { - await fh.close(); - } - break; - } catch (err) { - if ((err as NodeJS.ErrnoException).code === 'ENOENT') break; // nothing planted there - if (attempt >= 40) throw err; // ~6s of retries: report the leak - await new Promise((resolve) => setTimeout(resolve, 150)); - } - } - } -} - -/** - * Deterministic interrupted-checkpoint signature: build rows on a writable - * session with AUTO-CHECKPOINT DISABLED and close WITHOUT checkpointing, so — - * exactly like a CHECKPOINT killed mid-flight — the main file is stale and - * every row lives only in the WAL. Then rename that WAL to the - * `lbug.wal.checkpoint` name the engine gives it during checkpoint, and plant - * the shadow + intent/apply lock files it leaves behind. + * Preserve the main file and WAL before native close forces a checkpoint. + * Restoring both snapshots models a killed checkpoint with every row still + * WAL-only. All query results must be closed before releasing the database: + * native results retain the database and its writer lock until closed or GC'd. */ async function plantInterruptedCheckpoint(dbPath: string): Promise { - // Raw constructor (positional args mirror createLbugDatabase) because the - // autoCheckpoint toggle is not exposed through the config helpers — and - // auto-checkpoint-on-close is precisely what must NOT happen here. + // Disable automatic checkpoints while building the WAL. Native close still + // forces a checkpoint, so the pre-close main file must also be preserved. const db = new lbug.Database( dbPath, 128 * 1024 * 1024, // bufferManagerSize @@ -83,40 +53,61 @@ async function plantInterruptedCheckpoint(dbPath: string): Promise { false, // throwOnWalReplayFailure true, // enableChecksums ); - await db.init(); - const conn = new lbug.Connection(db); + let conn: lbug.Connection | undefined; + let mainBuffer: Buffer; + let walBuffer: Buffer; try { - await conn.query('CREATE NODE TABLE Person (name STRING, PRIMARY KEY(name))'); + await db.init(); + conn = new lbug.Connection(db); + const statements = ['CREATE NODE TABLE Person (name STRING, PRIMARY KEY(name))']; for (let i = 0; i < ROWS; i += 100) { const batch = Array.from({ length: 100 }, (_, j) => `{name: 'p${i + j}'}`).join(', '); - await conn.query(`UNWIND [${batch}] AS r CREATE (:Person {name: r.name})`); + statements.push(`UNWIND [${batch}] AS r CREATE (:Person {name: r.name})`); } - const walBuffer = await fs.readFile(`${dbPath}.wal`); - // Honesty check: the rows must actually LIVE in the WAL — on an engine - // that tolerates the planted state this is the only proof the plant is - // not an empty shell (review finding: unused walBuffer). + for (const statement of statements) { + const result = await conn.query(statement); + try { + for (const cursor of Array.isArray(result) ? result : [result]) await cursor.getAll(); + } finally { + await closeQueryResults(result); + } + } + [mainBuffer, walBuffer] = await Promise.all([ + fs.readFile(dbPath), + fs.readFile(`${dbPath}.wal`), + ]); expect(walBuffer.byteLength).toBeGreaterThan(0); - // Close WITHOUT checkpoint: rows stay WAL-only, main file stays stale. - // Explicitly awaited release BEFORE the rename/reopen — on Windows the - // kernel releases the engine's handles/locks asynchronously and the WAL - // rename + pooled reopen race them (Win32 Error 33, seen in CI). - await conn.close().catch(() => {}); + } finally { + await conn?.close().catch(() => {}); await db.close().catch(() => {}); - await waitForFixtureRelease(dbPath); - // Re-plant the captured WAL bytes rather than renaming the original: a - // close-time auto-checkpoint can consume the live .wal file out from - // under the rename (ENOENT — the fixture's other CI flake), while the - // captured buffer is what a killed checkpoint would have left behind. - await fs.writeFile(`${dbPath}.wal.checkpoint`, walBuffer); - await fs.writeFile(`${dbPath}.wal`, ''); - await fs.writeFile(`${dbPath}.shadow`, ''); - await fs.writeFile(`${dbPath}.checkpoint.intent.lock`, ''); - await fs.writeFile(`${dbPath}.checkpoint.apply.lock`, ''); - } catch (err) { - await conn.close().catch(() => {}); - await db.close().catch(() => {}); - throw err; } + // A separate process must acquire the native writer lock before the + // fixture files are restored. A byte-zero file read cannot test that lock. + const probe = spawnSync( + process.execPath, + [ + '-e', + `const lbug = require(process.argv[1]); + const db = new lbug.Database(process.argv[2], 128 * 1024 * 1024, false, + false, 16 * 1024 * 1024 * 1024); + db.init().then(() => db.close()).catch((error) => { + console.error(error); + process.exitCode = 1; + });`, + fileURLToPath(new URL('../../node_modules/@ladybugdb/core', import.meta.url)), + dbPath, + ], + { encoding: 'utf8', timeout: 15_000 }, + ); + expect(probe.error).toBeUndefined(); + expect(probe.status, probe.stderr).toBe(0); + + await fs.writeFile(dbPath, mainBuffer); + await fs.writeFile(`${dbPath}.wal.checkpoint`, walBuffer); + await fs.writeFile(`${dbPath}.wal`, ''); + await fs.writeFile(`${dbPath}.shadow`, ''); + await fs.writeFile(`${dbPath}.checkpoint.intent.lock`, ''); + await fs.writeFile(`${dbPath}.checkpoint.apply.lock`, ''); } describe('interrupted-checkpoint recovery (pooled read path self-heal)', () => { @@ -128,6 +119,38 @@ describe('interrupted-checkpoint recovery (pooled read path self-heal)', () => { if (tmpDir) await fs.rm(tmpDir, { recursive: true, force: true }); }); + it('preserves the setup error when both native closes reject', async () => { + const directory = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-lbug-cp-cleanup-')); + const setupError = new Error('injected query failure'); + const closeConnection = lbug.Connection.prototype.close; + const closeDatabase = lbug.Database.prototype.close; + const query = vi.spyOn(lbug.Connection.prototype, 'query').mockRejectedValueOnce(setupError); + const connectionClose = vi + .spyOn(lbug.Connection.prototype, 'close') + .mockImplementation(async function (this: lbug.Connection) { + await closeConnection.call(this); + throw new Error('injected connection close failure'); + }); + const databaseClose = vi + .spyOn(lbug.Database.prototype, 'close') + .mockImplementation(async function (this: lbug.Database) { + await closeDatabase.call(this); + throw new Error('injected database close failure'); + }); + try { + await expect(plantInterruptedCheckpoint(path.join(directory, 'lbug'))).rejects.toBe( + setupError, + ); + expect(connectionClose).toHaveBeenCalledOnce(); + expect(databaseClose).toHaveBeenCalledOnce(); + } finally { + query.mockRestore(); + connectionClose.mockRestore(); + databaseClose.mockRestore(); + await fs.rm(directory, { recursive: true, force: true }); + } + }); + it('opens read-only through the pool refusal and answers queries', async (ctx) => { // Engine-version honesty: only 0.19+ treats the planted signature as an // interrupted checkpoint ("Cannot open database in read-only mode while @@ -180,9 +203,6 @@ describe('interrupted-checkpoint recovery (pooled read path self-heal)', () => { } })(), ).rejects.toThrow(/checkpoint is in progress/i); - // The probe's native close is best-effort; its handles must be gone - // before the pooled open below (Windows Error 33 otherwise). - await waitForFixtureRelease(dbPath); // The wiki path: pooled READ-ONLY open. Before the fix this refused with // "Cannot open database in read-only mode while checkpoint is in @@ -212,7 +232,6 @@ describe('interrupted-checkpoint recovery (pooled read path self-heal)', () => { // A second open must answer without needing recovery again. await closeLbug(REPO); - await waitForFixtureRelease(dbPath); await initLbug(REPO, dbPath); const again = await executeQuery(REPO, 'MATCH (n:Person) RETURN count(n) AS c'); expect(again.length).toBe(1); diff --git a/gitnexus/test/integration/lbug-load-overlap-errors.test.ts b/gitnexus/test/integration/lbug-load-overlap-errors.test.ts index d72477a4e..f75c87e3a 100644 --- a/gitnexus/test/integration/lbug-load-overlap-errors.test.ts +++ b/gitnexus/test/integration/lbug-load-overlap-errors.test.ts @@ -81,6 +81,52 @@ afterAll(async () => { }); describe('loadGraphToLbug overlap error paths (#2226 F1)', () => { + it.each([1000, 0])( + 'retains COPY failures and rejects skipped rows with warning_limit=%i', + async (warningLimit) => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const { _captureLogger } = await import('../../src/core/logger.js'); + const graph = buildTestGraph([], []); + emitMock.mockImplementation( + async (_g: unknown, _r: unknown, dir: string, onNodes?: (n: NodeFiles) => void) => { + await fs.mkdir(dir, { recursive: true }); + const csvPath = path.join(dir, 'function.csv'); + await fs.writeFile( + csvPath, + 'id,name,filePath,startLine,endLine,isExported,content,description,convexEndpointFactory\n' + + '"Function:bad.ts:copy","copy","bad.ts",bad,2,false,"","",""\n', + ); + const nodeFiles = new Map([['Function', { csvPath, rows: 1 }]]) as NodeFiles; + onNodes?.(nodeFiles); + return { ...emptyResult(), nodeFiles }; + }, + ); + const capture = _captureLogger(); + try { + await adapter.executeQuery(`CALL warning_limit=${warningLimit}`); + await expect(adapter.loadGraphToLbug(graph, tmpBase, storagePath)).rejects.toThrow( + /skipped/i, + ); + const records = capture.records(); + expect(records.some((r) => /first COPY failure/i.test(String(r.msg)))).toBe(true); + expect(records).toContainEqual( + expect.objectContaining({ + expectedRows: 1, + copiedRows: 0, + skippedRows: 1, + retainedWarnings: warningLimit === 0 ? 0 : 1, + }), + ); + } finally { + try { + await adapter.executeQuery('CALL warning_limit=1000'); + } finally { + capture.restore(); + } + } + }, + ); + it('relationship-emit failure with node COPY in flight surfaces the emit error and leaks no unhandled rejection', async () => { const adapter = await import('../../src/core/lbug/lbug-adapter.js'); const graph = buildTestGraph( diff --git a/gitnexus/test/unit/api-graph-streaming.test.ts b/gitnexus/test/unit/api-graph-streaming.test.ts index 58de01f84..f79a1ea5a 100644 --- a/gitnexus/test/unit/api-graph-streaming.test.ts +++ b/gitnexus/test/unit/api-graph-streaming.test.ts @@ -1,4 +1,5 @@ import { EventEmitter } from 'node:events'; +import type { GraphRelationship } from 'gitnexus-shared'; import { describe, expect, it, vi, beforeEach } from 'vitest'; const { lbugMocks } = vi.hoisted(() => ({ @@ -23,6 +24,60 @@ const createMockResponse = (writeImpl?: (chunk: string) => boolean) => { return response; }; +describe('graph relationship reason contract', () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + it.each([ + { stored: null, expected: '' }, + { stored: undefined, expected: '' }, + { stored: '', expected: '' }, + { stored: 'branch → next', expected: 'branch → next' }, + ])( + 'returns reason=$expected for stored $stored in JSON and NDJSON', + async ({ stored, expected }) => { + const row = { + sourceId: 'BasicBlock:src/app.ts:1', + targetId: 'BasicBlock:src/app.ts:2', + type: 'CFG' as const, + confidence: 1, + reason: stored, + step: 0, + }; + lbugMocks.executeQuery.mockImplementation(async (query: string) => + query.includes('CodeRelation') ? [row] : [], + ); + lbugMocks.streamQuery.mockImplementation( + async (query: string, onRow: (row: unknown) => Promise) => { + if (!query.includes('CodeRelation')) return 0; + await onRow(row); + return 1; + }, + ); + + const buffered = JSON.parse(JSON.stringify(await buildGraph())); + const writes: string[] = []; + await streamGraphNdjson( + createMockResponse((chunk) => { + writes.push(chunk); + return true; + }), + ); + const streamed = writes.map((chunk) => JSON.parse(chunk)); + const relationship: GraphRelationship = { + ...row, + id: `${row.sourceId}_${row.type}_${row.targetId}`, + reason: expected, + }; + expect({ buffered: buffered.relationships, streamed }).toEqual({ + buffered: [relationship], + streamed: [{ type: 'relationship', data: relationship }], + }); + }, + ); +}); + describe('streamGraphNdjson', () => { beforeEach(() => { vi.clearAllMocks(); diff --git a/gitnexus/test/unit/basicblock-callee-ids-schema.test.ts b/gitnexus/test/unit/basicblock-callee-ids-schema.test.ts index a9e7ab1ad..d6dbb8154 100644 --- a/gitnexus/test/unit/basicblock-callee-ids-schema.test.ts +++ b/gitnexus/test/unit/basicblock-callee-ids-schema.test.ts @@ -74,7 +74,7 @@ describe('BasicBlock calleeIds — CSV header + row builder', () => { expect(rowCells).toHaveLength(headerCols.length); const calleeIdsIdx = headerCols.indexOf('calleeIds'); const calleesIdx = headerCols.indexOf('callees'); - // escapeCSVField always wraps the cell in double quotes; the space-joined + // Non-empty string fields are quoted; the space-joined // id list contains no comma, so the cell is a single CSV column. expect(rowCells[calleeIdsIdx]).toBe('"id1 id2"'); expect(rowCells[calleesIdx]).toBe('"foo bar"'); @@ -85,7 +85,8 @@ describe('BasicBlock calleeIds — CSV header + row builder', () => { const headerCols = BASICBLOCK_CSV_HEADER.split(','); const rowCells = buildBasicBlockRow(node).split(','); const calleeIdsIdx = headerCols.indexOf('calleeIds'); - expect(rowCells[calleeIdsIdx]).toBe('""'); + // An absent string must stay unquoted so native COPY loads SQL NULL. + expect(rowCells[calleeIdsIdx]).toBe(''); expect(rowCells[calleeIdsIdx]).not.toContain('undefined'); }); }); diff --git a/gitnexus/test/unit/csv-escaping.test.ts b/gitnexus/test/unit/csv-escaping.test.ts index 6abb31511..2e8d8a36a 100644 --- a/gitnexus/test/unit/csv-escaping.test.ts +++ b/gitnexus/test/unit/csv-escaping.test.ts @@ -15,16 +15,25 @@ import { // ─── escapeCSVField ────────────────────────────────────────────────── describe('escapeCSVField', () => { - it('returns empty quoted string for null', () => { - expect(escapeCSVField(null)).toBe('""'); + it('returns an unquoted NULL field for null', () => { + expect(escapeCSVField(null)).toBe(''); }); - it('returns empty quoted string for undefined', () => { - expect(escapeCSVField(undefined)).toBe('""'); + it('returns an unquoted NULL field for undefined', () => { + expect(escapeCSVField(undefined)).toBe(''); }); - it('returns quoted empty string for empty input', () => { - expect(escapeCSVField('')).toBe('""'); + it('preserves the legacy NULL meaning of empty input', () => { + expect(escapeCSVField('')).toBe(''); + }); + + it('returns a NULL field when sanitization removes the entire value', () => { + expect(escapeCSVField('\x00\x01')).toBe(''); + }); + + it('keeps whitespace and zero as quoted non-null values', () => { + expect(escapeCSVField(' ')).toBe('" "'); + expect(escapeCSVField(0)).toBe('"0"'); }); it('wraps simple string in quotes', () => { diff --git a/gitnexus/test/unit/detect-changes-format.test.ts b/gitnexus/test/unit/detect-changes-format.test.ts index 18f21eeeb..32e3a57f4 100644 --- a/gitnexus/test/unit/detect-changes-format.test.ts +++ b/gitnexus/test/unit/detect-changes-format.test.ts @@ -1,6 +1,7 @@ import { afterEach, beforeEach, describe, expect, it } from 'vitest'; import { formatDetectChangesResult } from '../../src/cli/detect-changes-format.js'; import { setCliLanguage } from '../../src/cli/i18n/index.js'; +import { parseDiffHunks } from '../../src/storage/git.js'; describe('formatDetectChangesResult — zero-symbol honesty (#3131)', () => { beforeEach(() => { @@ -62,6 +63,49 @@ describe('formatDetectChangesResult — zero-symbol honesty (#3131)', () => { expect(text).toBe('No changes detected.'); }); + it('explains unmapped source files without claiming a query failed or retry will repair the index', () => { + const text = formatDetectChangesResult({ + partial: true, + unmapped_files: ['src/index-lock.ts'], + summary: { changed_count: 0, affected_count: 0, changed_files: 1, risk_level: 'unknown' }, + }); + expect(text).toContain('PARTIAL RESULT'); + expect(text).toContain('src/index-lock.ts'); + expect(text).toMatch(/rebuild/i); + expect(text).not.toMatch(/queries failed|No changes detected|no indexed symbols overlap/i); + }); + + it('escapes Git-decoded terminal controls while leaving structured paths intact', () => { + const diff = [ + 'diff --git "a/evil\\033]52;c;VEVTVA==\\007.ts" "b/evil\\033]52;c;VEVTVA==\\007.ts"', + 'old mode 100644', + 'new mode 100755', + ].join('\n'); + const paths = parseDiffHunks(diff).map((file) => file.filePath); + expect(paths[0]).toContain('\u001b'); + const text = formatDetectChangesResult({ + partial: true, + unmapped_files: paths, + summary: { changed_count: 0, changed_files: 1, risk_level: 'unknown' }, + }); + expect(text).toContain('\\u001b]52;c;VEVTVA==\\u0007.ts'); + expect(text).not.toContain('\u001b'); + expect(text).not.toContain('\u0007'); + expect(paths[0]).toContain('\u0007'); + }); + + it('leads a populated summary with the incomplete source-mapping explanation', () => { + const text = formatDetectChangesResult({ + partial: true, + unmapped_files: ['other.ts'], + summary: { changed_count: 1, changed_files: 2, affected_count: 0, risk_level: 'unknown' }, + changed_symbols: [{ type: 'Function', name: 'known', filePath: 'code.py' }], + }); + expect(text.indexOf('other.ts')).toBeLessThan(text.indexOf('Changes:')); + expect(text).toContain('known'); + expect(text).toContain('unknown'); + }); + it('localizes the production clean-tree payload that carries English summary.message', () => { setCliLanguage('zh-CN'); const text = formatDetectChangesResult({ diff --git a/gitnexus/test/unit/detect-changes-hunk-scale.test.ts b/gitnexus/test/unit/detect-changes-hunk-scale.test.ts index a9e1035fc..b7f472dea 100644 --- a/gitnexus/test/unit/detect-changes-hunk-scale.test.ts +++ b/gitnexus/test/unit/detect-changes-hunk-scale.test.ts @@ -133,11 +133,11 @@ interface DetectChangesResult { partial?: boolean; } -async function runDetectChanges(): Promise { +async function runDetectChanges(scope = 'unstaged'): Promise { const backend = new LocalBackend(); await backend.init(); return (await backend.callTool('detect_changes', { - scope: 'unstaged', + scope, repo: 'hunk-scale-repo', })) as DetectChangesResult; } @@ -238,6 +238,135 @@ beforeEach(() => { }); describe('#2915 detect_changes hunk scaling', () => { + it('withholds a low-risk verdict when a changed source file maps to no symbols', async () => { + const result = await detectChangesForCodePy('def vanished():\n return 2\n'); + expect(result.summary.changed_files).toBe(1); + expect(result.summary.changed_count).toBe(0); + expect(result.summary.risk_level).toBe('unknown'); + expect(result.partial).toBe(true); + expect(result).toHaveProperty('unmapped_files', ['code.py']); + }); + + it.each(['notes.txt', 'README.md', 'config.json', 'config.yaml'])( + 'keeps ordinary non-source changes in %s measurable with zero symbols', + async (filePath) => { + const repoDir = makeRepo([filePath], 2); + writeFileSync(path.join(repoDir, filePath), 'changed\nline 2\n'); + registerRepo(repoDir); + const result = await runDetectChanges(); + expect(result.summary.risk_level).toBe('low'); + expect(result.partial).toBeUndefined(); + }, + ); + + it.each(['.html', '.htm', '.ejs', '.hbs', '.blade.php'])( + 'withholds ranked risk when a changed %s template maps to no symbols', + async (extension) => { + const filePath = `views/orders${extension}`; + const repoDir = makeRepo([filePath], 2); + writeFileSync(path.join(repoDir, filePath), '
\n'); + registerRepo(repoDir); + const result = await runDetectChanges(); + expect(result.summary).toMatchObject({ + changed_files: 1, + changed_count: 0, + risk_level: 'unknown', + }); + expect(result.partial).toBe(true); + expect(result).toHaveProperty('unmapped_files', [filePath]); + }, + ); + + it.each(['.html', '.htm', '.ejs', '.hbs', '.blade.php'])( + 'recognizes the template side of a pure %s-to-text rename', + async (extension) => { + const filePath = `orders${extension}`; + const repoDir = makeRepo([filePath], 2); + execFileSync('git', ['mv', filePath, 'orders.txt'], { cwd: repoDir }); + registerRepo(repoDir); + const result = await runDetectChanges('staged'); + expect(result.summary).toMatchObject({ + changed_files: 1, + changed_count: 0, + risk_level: 'unknown', + }); + expect(result.partial).toBe(true); + expect(result).toHaveProperty('unmapped_files', ['orders.txt']); + }, + ); + + it('keeps mapped template changes measurable', async () => { + const filePath = 'views/orders.blade.php'; + const repoDir = makeRepo([filePath], 2); + writeFileSync(path.join(repoDir, filePath), '
\nline 2\n'); + registerRepo(repoDir); + mockSymbolRows([{ name: 'orders', filePath, startLine: 0, endLine: 1 }]); + const result = await runDetectChanges(); + expect(result.summary).toMatchObject({ + changed_files: 1, + changed_count: 1, + risk_level: 'low', + }); + expect(result.changed_symbols.map((symbol) => symbol.name)).toEqual(['orders']); + expect(result.partial).toBeUndefined(); + expect(result).not.toHaveProperty('unmapped_files'); + }); + + it('withholds a low-risk verdict for an unmapped source rename without hunks', async () => { + const repoDir = makeRepo(['code.py'], 2); + execFileSync('git', ['mv', 'code.py', 'renamed.py'], { cwd: repoDir }); + registerRepo(repoDir); + const result = await runDetectChanges('staged'); + expect(result.summary.changed_files).toBe(1); + expect(result.summary.risk_level).toBe('unknown'); + expect(result.partial).toBe(true); + expect(result).toHaveProperty('unmapped_files', ['renamed.py']); + }); + + it('recognizes the source side of a pure source-to-text rename', async () => { + const repoDir = makeRepo(['code.py'], 2); + execFileSync('git', ['mv', 'code.py', 'code.txt'], { cwd: repoDir }); + registerRepo(repoDir); + const result = await runDetectChanges('staged'); + expect(result.summary).toMatchObject({ + changed_files: 1, + changed_count: 0, + risk_level: 'unknown', + }); + expect(result.partial).toBe(true); + expect(result).toHaveProperty('unmapped_files', ['code.txt']); + }); + + it.each(['.jcl', '.job', '.proc', '.copybook', '.JCL', '.COPYBOOK'])( + 'withholds ranked risk for an unmapped ingestion-supported %s rename', + async (extension) => { + const repoDir = makeRepo([`source${extension}`], 2); + execFileSync('git', ['mv', `source${extension}`, `renamed${extension}`], { cwd: repoDir }); + registerRepo(repoDir); + const result = await runDetectChanges('staged'); + expect(result.summary.risk_level).toBe('unknown'); + expect(result.partial).toBe(true); + expect(result).toHaveProperty('unmapped_files', [`renamed${extension}`]); + }, + ); + + it('retains mapped symbols while withholding ranked risk for an unmapped source', async () => { + const repoDir = makeRepo(['code.py', 'other.ts'], 2); + writeFileSync(path.join(repoDir, 'code.py'), 'changed\nline 2\n'); + writeFileSync(path.join(repoDir, 'other.ts'), 'changed\nline 2\n'); + registerRepo(repoDir); + mockSymbolRows([{ name: 'known', startLine: 0, endLine: 1 }]); + const result = await runDetectChanges(); + expect(result.summary).toMatchObject({ + changed_files: 2, + changed_count: 1, + risk_level: 'unknown', + }); + expect(result.changed_symbols.map((symbol) => symbol.name)).toEqual(['known']); + expect(result.partial).toBe(true); + expect(result).toHaveProperty('unmapped_files', ['other.ts']); + }); + it('sends the same query for a 3,000-hunk diff as for a 1-hunk diff', async () => { const oneHunkRepo = makeRepo(['big.txt'], 12000); editEveryNthLine(oneHunkRepo, 'big.txt', 12000, 12000); diff --git a/gitnexus/test/unit/eval-formatters.test.ts b/gitnexus/test/unit/eval-formatters.test.ts index bfc025432..d43ec5f38 100644 --- a/gitnexus/test/unit/eval-formatters.test.ts +++ b/gitnexus/test/unit/eval-formatters.test.ts @@ -641,7 +641,7 @@ describe('formatDetectChangesResult', () => { // counts at zero. Without the note the pre-commit gate reads as "clean". const result = formatDetectChangesResult({ partial: true, summary: { changed_count: 0 } }); expect(result).toContain('PARTIAL RESULT'); - expect(result).toContain('a graph query failed'); + expect(result).toContain('mapping is incomplete'); expect(result).not.toContain('No changes detected.'); }); diff --git a/gitnexus/test/unit/format-path.test.ts b/gitnexus/test/unit/format-path.test.ts new file mode 100644 index 000000000..a24446fe5 --- /dev/null +++ b/gitnexus/test/unit/format-path.test.ts @@ -0,0 +1,18 @@ +import { describe, expect, it } from 'vitest'; +import { formatPathForTerminal } from '../../src/cli/format-path.js'; + +describe('terminal path rendering', () => { + it('preserves ordinary Unicode and spaces', () => { + expect(formatPathForTerminal('src/王 name.ts')).toBe('src/王 name.ts'); + }); + + it.each(['\u001b', '\u0007', '\r', '\n', '\u007f', '\u0085', '\u009b'])( + 'renders control %j visibly without changing the decoded path', + (control) => { + const filePath = `src/a${control}.ts`; + const rendered = formatPathForTerminal(filePath); + expect(rendered).not.toMatch(/[\u0000-\u001f\u007f-\u009f]/); + expect(JSON.parse(rendered)).toBe(filePath); + }, + ); +}); diff --git a/gitnexus/test/unit/incremental-write-benchmark-verdict.test.ts b/gitnexus/test/unit/incremental-write-benchmark-verdict.test.ts new file mode 100644 index 000000000..e1c3b0c35 --- /dev/null +++ b/gitnexus/test/unit/incremental-write-benchmark-verdict.test.ts @@ -0,0 +1,59 @@ +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import vm from 'node:vm'; +import { parse } from '@babel/parser'; +import { isVariableDeclarator, traverseFast } from '@babel/types'; +import { expect, it } from 'vitest'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import { reconcileGraphNodeIdentities } from '../../src/core/incremental/write-reconciliation.js'; + +// Execute the standalone benchmark's actual audit closure, without its timed +// workload, against the production oracle and controlled native scan results. +const source = readFileSync( + new URL('../../bench/incremental-write-integrity/measure.cjs', import.meta.url), + 'utf8', +); +let auditSource: string | undefined; +traverseFast(parse(source, { sourceType: 'script' }), (node) => { + if ( + isVariableDeclarator(node) && + node.id.type === 'Identifier' && + node.id.name === 'audit' && + node.init?.start != null && + node.init.end != null + ) { + auditSource = source.slice(node.init.start, node.init.end); + } +}); +if (!auditSource) throw new Error('Benchmark audit closure not found'); + +const row = { id: 'Function:a.ts:f', name: 'f', filePath: 'a.ts', startLine: 0, endLine: 1 }; +it.each([ + { name: 'healthy scan', rows: [row], verdict: 'certified' }, + { name: 'complete tuple set plus duplicate row', rows: [row, row], verdict: 'rejected' }, + { + name: 'complete tuple set plus unexpected row', + rows: [row, { ...row, id: 'Function:a.ts:extra' }], + verdict: 'rejected', + }, + { name: 'missing tuple', rows: [], verdict: 'rejected' }, + { name: 'inconsistent range', rows: [{ ...row, startLine: 99 }], verdict: 'rejected' }, +])('reports the production rejection verdict for $name', async ({ rows, verdict }) => { + const graph = createKnowledgeGraph(); + graph.addNode({ id: row.id, label: 'Function', properties: { ...row } }); + const wanted = new Set([ + JSON.stringify([row.id, row.name, row.filePath, row.startLine, row.endLine]), + ]); + const nativePhases: { verdict: string }[] = []; + const audit = vm.runInNewContext(`(${auditSource})`, { + assert, + graph, + wanted, + nativePhases, + size: { nodes: 1 }, + reconcileGraphNodeIdentities, + adapter: { executeQuery: async (query: string) => (query.includes('Function') ? rows : []) }, + }) as (phase: string) => Promise; + await audit('controlled-scan'); + expect(nativePhases).toEqual([expect.objectContaining({ verdict, rows: rows.length })]); +}); diff --git a/gitnexus/test/unit/incremental-write-integrity.test.ts b/gitnexus/test/unit/incremental-write-integrity.test.ts new file mode 100644 index 000000000..14ac9fca9 --- /dev/null +++ b/gitnexus/test/unit/incremental-write-integrity.test.ts @@ -0,0 +1,267 @@ +import { appendFile } from 'node:fs/promises'; +import path from 'node:path'; +import { afterEach, describe, expect, it, vi } from 'vitest'; +import { setupMiniRepo } from '../helpers/mini-repo.js'; +import { getStoragePaths, loadMeta, readRegistry } from '../../src/storage/repo-manager.js'; +import * as adapter from '../../src/core/lbug/lbug-adapter.js'; +import * as fts from '../../src/core/search/fts-indexes.js'; +import * as analyzerIdentity from '../../src/core/analyzer-identity.js'; +import * as checkpoints from '../../src/core/lbug/wal-checkpoint-driver.js'; +import { commitAll, initGitRepo } from '../helpers/temp-git-repo.js'; +import { runFullAnalysis } from '../../src/core/run-analyze.js'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import { reconcileGraphNodeIdentities } from '../../src/core/incremental/write-reconciliation.js'; + +afterEach(() => { + vi.restoreAllMocks(); + vi.unstubAllEnvs(); +}); + +const options = { skipAgentsMd: true, skipSkills: true }; +const callbacks = { onProgress: () => {} }; + +async function touchHandler(repoPath: string): Promise { + initGitRepo(repoPath); + await appendFile(path.join(repoPath, 'src/handler.ts'), '\n// integrity probe\n'); + commitAll(repoPath, 'touch handler'); +} + +describe('incremental graph identity before publication', () => { + it('chooses a full write before selective mutation when manual checkpoints are disabled', async () => { + const repo = await setupMiniRepo('gitnexus-test-checkpoint-opt-out-'); + const { storagePath } = getStoragePaths(repo.dbPath); + try { + await runFullAnalysis(repo.dbPath, options, callbacks); + await touchHandler(repo.dbPath); + vi.stubEnv('GITNEXUS_WAL_MANUAL_CHECKPOINT', '0'); + const deleteNodes = vi.spyOn(adapter, 'deleteNodesForFiles'); + const flush = vi.spyOn(adapter, 'tryFlushWAL'); + const logs: string[] = []; + const result = await runFullAnalysis(repo.dbPath, options, { + ...callbacks, + onLog: (line) => logs.push(line), + }); + expect(deleteNodes).not.toHaveBeenCalled(); + expect(flush).not.toHaveBeenCalled(); + expect(result.incrementalStats).toBeUndefined(); + expect(logs).toContain( + 'Manual WAL checkpoints are disabled; switching to a full DB write before mutation.', + ); + expect((await loadMeta(storagePath))?.incrementalInProgress).toBeUndefined(); + } finally { + await adapter.closeLbug(); + await repo.cleanup(); + } + }); + + it('retries a transient checkpoint I/O failure at the final publication gate', async () => { + const repo = await setupMiniRepo('gitnexus-test-final-checkpoint-retry-'); + const { storagePath } = getStoragePaths(repo.dbPath); + try { + await runFullAnalysis(repo.dbPath, options, callbacks); + await touchHandler(repo.dbPath); + // Isolate the final gate from the periodic driver's independent cadence. + vi.spyOn(checkpoints, 'startWalCheckpointDriver').mockReturnValue({ stop: async () => {} }); + const flush = adapter.tryFlushWAL; + const build = fts.buildSearchIndexesOrDegrade; + let attempts = 0; + vi.spyOn(fts, 'buildSearchIndexesOrDegrade').mockImplementation(async (...args) => { + const result = await build(...args); + vi.spyOn(adapter, 'tryFlushWAL').mockImplementation(async () => { + if (++attempts === 1) { + throw new Error( + 'Runtime exception: IO exception: Error renaming file db.wal to db.wal.checkpoint', + ); + } + return flush(); + }); + return result; + }); + const result = await runFullAnalysis(repo.dbPath, options, callbacks); + expect(attempts).toBe(2); + expect(result.incrementalStats?.writeMode).toBe('incremental'); + expect((await loadMeta(storagePath))?.incrementalInProgress).toBeUndefined(); + } finally { + await adapter.closeLbug(); + await repo.cleanup(); + } + }); + it('matches native CSV normalization for nullable, absent-range and Unicode identity fields', async () => { + const repo = await setupMiniRepo('gitnexus-test-identity-parity-'); + const { storagePath, lbugPath } = getStoragePaths(repo.dbPath); + const graph = createKnowledgeGraph(); + graph.addNode({ + id: 'File:src/Å.ts', + label: 'File', + properties: { name: 'Å.ts', filePath: 'src/Å.ts' }, + }); + graph.addNode({ + id: 'Route:/王', + label: 'Route', + properties: { name: '/王', filePath: 'src/Å.ts' }, + }); + graph.addNode({ + id: 'Tool:王', + label: 'Tool', + properties: { name: '王', filePath: 'src/Å.ts' }, + }); + graph.addNode({ id: 'Destination:topic', label: 'Destination', properties: { name: 'topic' } }); + graph.addNode({ + id: 'Function:src/Å.ts:王\u0000\uD800', + label: 'Function', + properties: { name: '王\r\n\uD800', filePath: 'src/Å.ts' }, + }); + try { + await adapter.initLbug(lbugPath, { skipFts: true }); + await adapter.loadGraphToLbug( + graph, + repo.dbPath, + storagePath, + undefined, + undefined, + undefined, + 'none', + ); + await adapter.tryFlushWAL(); + await expect( + reconcileGraphNodeIdentities(graph, adapter.executeQuery, 'native CSV parity'), + ).resolves.toMatchObject({ nodes: 5 }); + } finally { + await adapter.closeLbug(); + await repo.cleanup(); + } + }); + + it.each([ + { phase: 'after-copy', atomicIncremental: false }, + { phase: 'after-fts', atomicIncremental: false }, + { phase: 'after-fts', atomicIncremental: true }, + { phase: 'after-fts-finalization', atomicIncremental: false }, + { phase: 'checkpoint-false', atomicIncremental: false }, + { phase: 'checkpoint-throw', atomicIncremental: true }, + ])( + 'refuses freshness after $phase failure (atomic=$atomicIncremental)', + async ({ phase, atomicIncremental }) => { + const repo = await setupMiniRepo('gitnexus-test-write-integrity-'); + const { storagePath, lbugPath } = getStoragePaths(repo.dbPath); + try { + const healthy = await runFullAnalysis(repo.dbPath, options, callbacks); + const before = await loadMeta(storagePath); + const registeredBefore = (await readRegistry()).find((entry) => entry.path === repo.dbPath); + const logs: string[] = []; + const victim = [...healthy.pipelineResult.graph.iterNodes()].find( + (node) => node.label === 'Function' && node.properties.filePath === 'src/validator.ts', + ); + expect(victim?.id).toBeTruthy(); + + await touchHandler(repo.dbPath); + const damage = async () => { + await adapter.executePrepared('MATCH (n:Function {id: $id}) DETACH DELETE n', { + id: victim.id, + }); + }; + if (phase === 'after-copy') { + const load = adapter.loadGraphToLbug; + vi.spyOn(adapter, 'loadGraphToLbug').mockImplementation(async (...args) => { + const result = await load(...args); + await damage(); + return result; + }); + } else { + const build = fts.buildSearchIndexesOrDegrade; + vi.spyOn(fts, 'buildSearchIndexesOrDegrade').mockImplementation(async (...args) => { + const result = await build(...args); + if (phase === 'checkpoint-false') { + vi.spyOn(adapter, 'tryFlushWAL').mockResolvedValue(false); + } else if (phase === 'checkpoint-throw') { + vi.spyOn(adapter, 'tryFlushWAL').mockRejectedValue( + new Error('injected final checkpoint failure'), + ); + } else { + await damage(); + if (phase === 'after-fts-finalization') { + vi.spyOn(analyzerIdentity, 'finalizeAnalyzerRunnerIdentity').mockImplementation( + () => { + throw new Error('injected analyzer finalization failure'); + }, + ); + } + } + return result; + }); + } + + await expect( + runFullAnalysis( + repo.dbPath, + { ...options, atomicIncremental }, + { + ...callbacks, + onLog: (line) => logs.push(line), + }, + ), + ).rejects.toThrow( + phase.startsWith('checkpoint') + ? /checkpoint/i + : phase === 'after-fts-finalization' + ? /injected analyzer finalization failure/ + : /graph.*reconciliation/i, + ); + const after = await loadMeta(storagePath); + expect(after?.lastCommit).toBe(before?.lastCommit); + expect(after?.indexedAt).toBe(before?.indexedAt); + expect(logs.filter((line) => line.includes('atomic-incremental'))).toEqual( + atomicIncremental ? [expect.stringContaining('staged')] : [], + ); + expect(Boolean(after?.incrementalInProgress)).toBe(!atomicIncremental); + const registeredAfter = (await readRegistry()).find((entry) => entry.path === repo.dbPath); + expect(registeredAfter?.indexedAt).toBe(registeredBefore?.indexedAt); + expect(registeredAfter?.lastCommit).toBe(registeredBefore?.lastCommit); + vi.restoreAllMocks(); + if (atomicIncremental) { + await adapter.initLbug(lbugPath); + const retained = await adapter.executePrepared( + 'MATCH (n:Function {id: $id}) RETURN n.id AS id', + { + id: victim.id, + }, + ); + expect(retained).toEqual([{ id: victim.id }]); + await adapter.closeLbug(); + } else { + expect(after?.incrementalInProgress).toMatchObject({ + phase: 'graph-reconciliation', + checkpointSucceeded: false, + }); + await expect( + runFullAnalysis(repo.dbPath, { ...options, repairFts: true }, callbacks), + ).rejects.toThrow(/mid-incremental-recovery/); + expect((await loadMeta(storagePath))?.incrementalInProgress?.phase).toBe( + 'graph-reconciliation', + ); + // The dirty marker drives recovery without a manual --force, and a + // no-op rerun after recovery must stop rebuilding. + const recovered = await runFullAnalysis(repo.dbPath, options, callbacks); + expect(recovered.incrementalStats).toBeUndefined(); + expect((await loadMeta(storagePath))?.incrementalInProgress).toBeUndefined(); + await adapter.initLbug(lbugPath, { readOnly: true }); + await expect( + reconcileGraphNodeIdentities( + recovered.pipelineResult.graph, + adapter.executeQuery, + 'recovered/reopened', + ), + ).resolves.toMatchObject({ nodes: expect.any(Number) }); + await adapter.closeLbug(); + expect((await runFullAnalysis(repo.dbPath, options, callbacks)).alreadyUpToDate).toBe( + true, + ); + } + } finally { + await adapter.closeLbug(); + await repo.cleanup(); + } + }, + 120_000, + ); +}); diff --git a/gitnexus/test/unit/incremental-write-reproducer-cleanup.test.ts b/gitnexus/test/unit/incremental-write-reproducer-cleanup.test.ts new file mode 100644 index 000000000..ed3f2dbfe --- /dev/null +++ b/gitnexus/test/unit/incremental-write-reproducer-cleanup.test.ts @@ -0,0 +1,265 @@ +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import vm from 'node:vm'; +import { expect, it } from 'vitest'; + +const source = readFileSync( + new URL('../../bench/incremental-write-integrity/reproduce.cjs', import.meta.url), + 'utf8', +); + +type FailurePoint = + | 'write:1' + | 'write:2' + | 'database:1' + | 'database:2' + | 'connection:1' + | 'connection:2' + | 'connection-close:1' + | 'connection-close:2' + | 'database-close:1' + | 'database-close:2' + | 'remove'; + +// Execute the actual entry point with controlled native and file APIs. Healthy +// scan tuples allow constructor and cleanup faults in both native sessions. +async function runReproducer( + failures: FailurePoint[] = [], + keep = false, + options: { + requireCorruption?: boolean; + copiedScan?: 'missing' | 'duplicate' | 'wrong-field'; + } = {}, +) { + const directory = path.join(os.tmpdir(), 'ladybug-copy-identity-controlled'); + const events: string[] = []; + const diagnostics: unknown[] = []; + const injected = new Map(failures.map((point) => [point, new Error(point)])); + let databases = 0; + let connections = 0; + let writes = 0; + let scans = 0; + let indices = Array.from({ length: 8192 }, (_, i) => i); + const output: string[] = []; + const processDouble = { + argv: [ + 'node', + 'reproduce.cjs', + ...(keep ? ['--keep'] : []), + ...(options.requireCorruption ? ['--require-corruption'] : []), + ], + version: process.version, + exitCode: 0, + stdout: { write: (value: string) => output.push(value) }, + }; + const fail = (point: FailurePoint) => { + const error = injected.get(point); + if (error) throw error; + }; + class Database { + readonly number = ++databases; + constructor() { + events.push(`database:${this.number}`); + fail(`database:${this.number}` as FailurePoint); + } + async close() { + events.push(`database-close:${this.number}`); + fail(`database-close:${this.number}` as FailurePoint); + } + } + class Connection { + readonly number = ++connections; + constructor(database: Database) { + events.push(`connection:${this.number}`); + expect(database.number).toBe(this.number); + fail(`connection:${this.number}` as FailurePoint); + } + async query(cypher: string) { + if (cypher.includes('DETACH DELETE')) { + indices = indices.filter((i) => i >= 32 && i < 8192 - 32); + } else if (cypher.startsWith('COPY')) { + indices = Array.from({ length: 8192 }, (_, i) => i); + } + const rows = cypher.includes('RETURN id(n)') + ? indices.map((i) => ({ + internalID: { offset: i }, + id: `Function:src/owner${Math.floor(i / 32)}.ts:fn${i}`, + name: `fn${i}`, + filePath: `src/owner${Math.floor(i / 32)}.ts`, + startLine: (i % 32) * 4, + endLine: (i % 32) * 4 + 2, + })) + : []; + if (cypher.includes('RETURN id(n)') && ++scans === 3) { + if (options.copiedScan === 'missing') rows.pop(); + if (options.copiedScan === 'duplicate') rows.push({ ...rows[0] }); + if (options.copiedScan === 'wrong-field') rows[0].name = 'wrong-function'; + } + return { getAll: async () => rows, close: async () => {} }; + } + async close() { + events.push(`connection-close:${this.number}`); + fail(`connection-close:${this.number}` as FailurePoint); + } + } + const fsDouble = { + mkdtemp: async () => directory, + writeFile: async () => { + const point = `write:${++writes}` as FailurePoint; + events.push(point); + fail(point); + }, + rm: async (target: string, options: { recursive: boolean; force: boolean }) => { + expect(target).toBe(directory); + expect(options).toEqual({ recursive: true, force: true }); + events.push('remove'); + fail('remove'); + }, + }; + const modules: Record = { + 'node:assert/strict': assert, + 'node:fs/promises': fsDouble, + 'node:os': os, + 'node:path': path, + '@ladybugdb/core': { Database, Connection, VERSION: 'controlled' }, + }; + await vm.runInNewContext(source, { + require: (name: string) => { + if (!(name in modules)) throw new Error(`Unexpected require: ${name}`); + return modules[name]; + }, + process: processDouble, + console: { error: (error: unknown) => diagnostics.push(error) }, + AggregateError, + }); + return { events, diagnostics, injected, output, exitCode: processDouble.exitCode, directory }; +} + +it.each([ + { point: 'write:1', closed: [] }, + { point: 'write:2', closed: [] }, + { point: 'database:1', closed: [] }, + { point: 'connection:1', closed: ['database-close:1'] }, + { point: 'database:2', closed: ['connection-close:1', 'database-close:1'] }, + { + point: 'connection:2', + closed: ['connection-close:1', 'database-close:1', 'database-close:2'], + }, + { point: 'connection-close:1', closed: ['connection-close:1', 'database-close:1'] }, + { point: 'database-close:1', closed: ['connection-close:1', 'database-close:1'] }, + { + point: 'connection-close:2', + closed: ['connection-close:1', 'database-close:1', 'connection-close:2', 'database-close:2'], + }, + { + point: 'database-close:2', + closed: ['connection-close:1', 'database-close:1', 'connection-close:2', 'database-close:2'], + }, +] satisfies { point: FailurePoint; closed: string[] }[])( + 'closes acquired resources and removes the directory when $point fails', + async ({ point, closed }) => { + const result = await runReproducer([point]); + expect(result.exitCode).toBe(1); + expect(result.diagnostics).toEqual([result.injected.get(point)]); + expect(result.events.filter((event) => event.includes('-close:'))).toEqual(closed); + expect(result.events.at(-1)).toBe('remove'); + expect(result.events.filter((event) => event === 'remove')).toHaveLength(1); + }, +); + +it('closes both native sessions and removes the directory after a healthy run', async () => { + const result = await runReproducer(); + expect(result.exitCode).toBe(0); + expect(result.diagnostics).toEqual([]); + expect(result.events.slice(-5)).toEqual([ + 'database:2', + 'connection:2', + 'connection-close:2', + 'database-close:2', + 'remove', + ]); + const report = JSON.parse(result.output.join('')); + expect(report.phases).toEqual([ + { phase: 'baseline', rows: 8192, wrong_tuples: 0, missing_tuples: 0, examples: [] }, + { phase: 'after-delete', rows: 8128, wrong_tuples: 0, missing_tuples: 64, examples: [] }, + { phase: 'after-copy', rows: 8192, wrong_tuples: 0, missing_tuples: 0, examples: [] }, + { phase: 'after-checkpoint', rows: 8192, wrong_tuples: 0, missing_tuples: 0, examples: [] }, + { phase: 'reopened', rows: 8192, wrong_tuples: 0, missing_tuples: 0, examples: [] }, + ]); + expect(report.directory).toBeUndefined(); +}); + +it.each([ + { copiedScan: 'missing', rows: 8191, wrong_tuples: 0, missing_tuples: 1 }, + { copiedScan: 'duplicate', rows: 8193, wrong_tuples: 0, missing_tuples: 0 }, + { copiedScan: 'wrong-field', rows: 8192, wrong_tuples: 1, missing_tuples: 1 }, +] as const)( + 'accepts a $copiedScan scan discrepancy as reproduced corruption', + async ({ copiedScan, ...expected }) => { + const result = await runReproducer([], false, { requireCorruption: true, copiedScan }); + const report = JSON.parse(result.output.join('')); + expect( + report.phases.find((phase: { phase: string }) => phase.phase === 'after-copy'), + ).toMatchObject(expected); + expect(result.exitCode).toBe(0); + expect(result.diagnostics).toEqual([]); + expect(result.events.at(-1)).toBe('remove'); + }, +); + +it('rejects --require-corruption when the native scan is healthy', async () => { + const result = await runReproducer([], false, { requireCorruption: true }); + expect(result.exitCode).toBe(1); + expect(result.diagnostics).toHaveLength(1); + expect((result.diagnostics[0] as Error).message).toContain('Native failure did not reproduce'); + expect(result.events.at(-1)).toBe('remove'); +}); + +it('keeps the directory deliberately without retaining native handles for --keep', async () => { + const result = await runReproducer([], true); + expect(result.exitCode).toBe(0); + expect(result.events).not.toContain('remove'); + expect(result.events.slice(-2)).toEqual(['connection-close:2', 'database-close:2']); + expect(JSON.parse(result.output.join('')).directory).toBe(result.directory); +}); + +it('honors --keep when connection construction fails', async () => { + const result = await runReproducer(['connection:1'], true); + expect(result.exitCode).toBe(1); + expect(result.diagnostics).toEqual([result.injected.get('connection:1')]); + expect(result.events.at(-1)).toBe('database-close:1'); + expect(result.events).not.toContain('remove'); +}); + +it('retains both close failures and still removes the directory', async () => { + const result = await runReproducer(['connection-close:1', 'database-close:1']); + expect(result.exitCode).toBe(1); + const error = result.diagnostics[0] as AggregateError; + expect(error).toBeInstanceOf(AggregateError); + expect(error.errors).toEqual([ + result.injected.get('connection-close:1'), + result.injected.get('database-close:1'), + ]); + expect(result.events.at(-1)).toBe('remove'); +}); + +it('retains the original construction error when database cleanup also fails', async () => { + const result = await runReproducer(['connection:1', 'database-close:1']); + expect(result.exitCode).toBe(1); + const error = result.diagnostics[0] as AggregateError; + expect(error).toBeInstanceOf(AggregateError); + expect(error.errors).toEqual([ + result.injected.get('connection:1'), + result.injected.get('database-close:1'), + ]); + expect(result.events.at(-1)).toBe('remove'); +}); + +it('reports directory-removal failures after closing the native handles', async () => { + const result = await runReproducer(['remove']); + expect(result.exitCode).toBe(1); + expect(result.diagnostics).toEqual([result.injected.get('remove')]); + expect(result.events.slice(-3)).toEqual(['connection-close:2', 'database-close:2', 'remove']); +}); diff --git a/gitnexus/test/unit/parse-diff-hunks.test.ts b/gitnexus/test/unit/parse-diff-hunks.test.ts index 0057803f5..84ba9ad03 100644 --- a/gitnexus/test/unit/parse-diff-hunks.test.ts +++ b/gitnexus/test/unit/parse-diff-hunks.test.ts @@ -164,7 +164,30 @@ describe('parseDiffHunks', () => { 'rename from src/old.ts', 'rename to src/new.ts', ].join('\n'); - expect(parseDiffHunks(diff)).toEqual([{ filePath: 'src/new.ts', hunks: [] }]); + expect(parseDiffHunks(diff)).toEqual([ + { filePath: 'src/new.ts', oldFilePath: 'src/old.ts', hunks: [] }, + ]); + }); + + it('decodes the source of a C-quoted rename and preserves it through content headers', () => { + const diff = [ + 'diff --git "a/src/old\\040name.py" b/src/new.txt', + 'similarity index 80%', + 'rename from "src/old\\040name.py"', + 'rename to src/new.txt', + '--- "a/src/old\\040name.py"', + '+++ b/src/new.txt', + '@@ -1 +1 @@', + '-old', + '+new', + ].join('\n'); + expect(parseDiffHunks(diff)).toEqual([ + { + filePath: 'src/new.txt', + oldFilePath: 'src/old name.py', + hunks: [{ startLine: 1, endLine: 1 }], + }, + ]); }); it('keeps line ranges for whitespace-only hunks', () => { @@ -241,7 +264,9 @@ describe('parseDiffHunks', () => { 'rename from plain.ts', 'rename to foo b/plain.ts', ].join('\n'); - expect(parseDiffHunks(diff)).toEqual([{ filePath: 'foo b/plain.ts', hunks: [] }]); + expect(parseDiffHunks(diff)).toEqual([ + { filePath: 'foo b/plain.ts', oldFilePath: 'plain.ts', hunks: [] }, + ]); }); it('keeps one FileDiff when a content line repeats +++ b/', () => { @@ -278,7 +303,7 @@ describe('parseDiffHunks', () => { 'rename to new.ts', ].join('\n'); expect(parseDiffHunksResult(diff)).toEqual({ - files: [{ filePath: 'new.ts', hunks: [] }], + files: [{ filePath: 'new.ts', oldFilePath: 'old name.ts', hunks: [] }], unparsedGitHeaders: 0, }); }); @@ -291,7 +316,7 @@ describe('parseDiffHunks', () => { 'rename to new name.ts', ].join('\n'); expect(parseDiffHunksResult(diff)).toEqual({ - files: [{ filePath: 'new name.ts', hunks: [] }], + files: [{ filePath: 'new name.ts', oldFilePath: 'old.ts', hunks: [] }], unparsedGitHeaders: 0, }); }); diff --git a/gitnexus/test/unit/run-analyze-fts-crash-marker.test.ts b/gitnexus/test/unit/run-analyze-fts-crash-marker.test.ts index 6997a0604..b372fb8dd 100644 --- a/gitnexus/test/unit/run-analyze-fts-crash-marker.test.ts +++ b/gitnexus/test/unit/run-analyze-fts-crash-marker.test.ts @@ -113,7 +113,13 @@ const mockLbugAdapter = async () => { initLbug: vi.fn(async () => undefined), loadGraphToLbug: vi.fn(async () => undefined), getLbugStats: vi.fn(async () => ({ nodes: 1, edges: 0, communities: 0, processes: 0 })), - executeQuery: vi.fn(async () => []), + // The policy fixture has one stored File row. Publication now reconciles + // that identity instead of trusting the synthetic node count alone. + executeQuery: vi.fn(async (query: string) => + query.startsWith('MATCH (n:`File`) RETURN n.id AS id') + ? [{ id: 'file:src/a.ts', name: '', filePath: REL_FILE }] + : [], + ), executeWithReusedStatement: vi.fn(async () => []), closeLbug: vi.fn(async () => undefined), wipeLbugDbFiles: vi.fn(async () => undefined), @@ -1011,9 +1017,12 @@ describe('runFullAnalysis FTS crash marker', () => { ); it('treats a boundary checkpoint failure as best-effort on an in-place plan', async () => { - const checkpointOnce = vi.fn(async () => { - throw new Error('checkpoint rename failed'); - }); + // Only the earlier boundary is best-effort. Publication still requires + // its own successful checkpoint through the same policy helper. + const checkpointOnce = vi + .fn<() => Promise>() + .mockRejectedValueOnce(new Error('checkpoint rename failed')) + .mockResolvedValue(true); let stamped: RepoMeta['incrementalInProgress']; vi.doMock('../../src/core/lbug/wal-checkpoint-driver.js', async (importActual) => ({ ...(await importActual()), @@ -1060,6 +1069,7 @@ describe('runFullAnalysis FTS crash marker', () => { { onProgress: () => {}, onLog: () => {} }, ); expect(result.ftsSkipped).not.toBe(true); + expect(checkpointOnce).toHaveBeenCalledTimes(2); expect(stamped).toMatchObject({ phase: FTS_DIRTY_PHASE, writePlan: 'in-place', diff --git a/gitnexus/test/unit/write-reconciliation.test.ts b/gitnexus/test/unit/write-reconciliation.test.ts new file mode 100644 index 000000000..d4d4a86ea --- /dev/null +++ b/gitnexus/test/unit/write-reconciliation.test.ts @@ -0,0 +1,90 @@ +import { describe, expect, it, vi } from 'vitest'; +import { reconcileGraphNodeIdentities } from '../../src/core/incremental/write-reconciliation.js'; +import { buildTestGraph } from '../helpers/test-graph.js'; + +const graph = () => + buildTestGraph( + [ + { + id: 'Function:src/lock.ts:acquire', + label: 'Function', + name: 'acquire', + filePath: 'src/lock.ts', + startLine: 504, + endLine: 719, + }, + { + id: 'Function:bench/measure.mjs:retainedHeapBytes', + label: 'Function', + name: 'retainedHeapBytes', + filePath: 'bench/measure.mjs', + startLine: 437, + endLine: 453, + }, + ], + [], + ); +const rows = () => [ + { + id: 'Function:src/lock.ts:acquire', + name: 'acquire', + filePath: 'src/lock.ts', + startLine: 504, + endLine: 719, + }, + { + id: 'Function:bench/measure.mjs:retainedHeapBytes', + name: 'retainedHeapBytes', + filePath: 'bench/measure.mjs', + startLine: 437, + endLine: 453, + }, +]; +const queryFor = (functions: ReturnType) => + vi.fn(async (query: string) => (query.includes('(n:`Function`)') ? functions : [])); + +describe('incremental identity reconciliation', () => { + it('certifies every identity using one unfiltered query per label', async () => { + const query = queryFor(rows()); + const receipt = await reconcileGraphNodeIdentities(graph(), query, 'post-COPY'); + expect(receipt.nodes).toBe(2); + expect(query).toHaveBeenCalledTimes(receipt.tables); + for (const [cypher] of query.mock.calls) { + expect(cypher).toMatch(/^MATCH \(n:`\w+`\) RETURN/); + expect(cypher).not.toMatch(/WHERE|content/); + } + }); + + it('rejects a missing unchanged symbol', async () => { + await expect( + reconcileGraphNodeIdentities(graph(), queryFor(rows().slice(1)), 'post-COPY'), + ).rejects.toThrow(/1 missing ID.*src\/lock.ts:acquire/); + }); + + it.each(['name', 'filePath', 'startLine', 'endLine'] as const)( + 'rejects column/row inconsistency in %s despite identical ID counts', + async (field) => { + const persisted = rows(); + const swapped = persisted.map((row, index) => ({ + ...row, + [field]: persisted[1 - index][field], + })); + await expect( + reconcileGraphNodeIdentities(graph(), queryFor(swapped), 'pre-publish'), + ).rejects.toThrow(field); + }, + ); + + it('rejects a duplicated row that would hide a missing ID in a count check', async () => { + await expect( + reconcileGraphNodeIdentities(graph(), queryFor([rows()[0], rows()[0]]), 'post-COPY'), + ).rejects.toThrow(/duplicate ID/); + }); + + it('surfaces a failed scan rather than certifying an unreadable graph', async () => { + const query = vi.fn().mockRejectedValue(new Error('Invalid UTF-8')); + await expect(reconcileGraphNodeIdentities(graph(), query, 'pre-publish')).rejects.toThrow( + /could not read identity fields.*Invalid UTF-8/, + ); + }); +}); diff --git a/gitnexus/vendor/lbug-fts/manifest.json b/gitnexus/vendor/lbug-fts/manifest.json index 6d93ddd03..62c4da0c7 100644 --- a/gitnexus/vendor/lbug-fts/manifest.json +++ b/gitnexus/vendor/lbug-fts/manifest.json @@ -1,6 +1,6 @@ { - "coreVersion": "0.18.3", - "extensionVersion": "0.18.1", + "coreVersion": "0.21.1", + "extensionVersion": "0.21.0", "filename": "libfts.lbug_extension", "officialRepo": "https://extension.ladybugdb.com/", "urlTemplate": "{officialRepo}v{extensionVersion}/{upstreamPlatform}/fts/{filename}", diff --git a/gitnexus/vendor/lbug-fts/prebuilds/SHA256SUMS b/gitnexus/vendor/lbug-fts/prebuilds/SHA256SUMS index 3aa968510..9683a3704 100644 --- a/gitnexus/vendor/lbug-fts/prebuilds/SHA256SUMS +++ b/gitnexus/vendor/lbug-fts/prebuilds/SHA256SUMS @@ -1,5 +1,5 @@ -01cf0d4debae1b0c7c182bf78747ee1b148a76667bbaa5bb0cb3cc848998028d ./win32-x64/libfts.lbug_extension -4af2602007a4a02d5b18d0b6ecb20b1cc2f493242a915faf7df1c033136107b1 ./linux-arm64/libfts.lbug_extension -cf687fda0f82bfbe116e775f2f17c39faf45be6300fd1937c2b1afd21142c121 ./linux-x64/libfts.lbug_extension -d3564c05ec28a0293a5488eaff2c06591624ac01a44af183d8cfacd92a29d785 ./darwin-arm64/libfts.lbug_extension -d8589c0a91c6667e959da6a81f712f69b330d0563602933da983edeebee60fc6 ./darwin-x64/libfts.lbug_extension +248332de781a7781f612a55b5397e88f58f1f1ba2a973836ada199eab0f6bf10 ./darwin-arm64/libfts.lbug_extension +742f2756c2f80bcd1886b6ee0446483038cff0ecf6cf47f698bc6fe855cbaef6 ./linux-x64/libfts.lbug_extension +7b6132ae6a554e3eeb3c4b0da7e2eb5c483b258b31e848a0e3d6000ac29885ee ./win32-x64/libfts.lbug_extension +b8675b086664e3d6742330a517c5b619f0ac10dd094d250f12bef9822a8d5750 ./linux-arm64/libfts.lbug_extension +c86f427503555a3ccdc55eb66a8d82793a206aac599a6092f7d1d4eb50c21403 ./darwin-x64/libfts.lbug_extension diff --git a/gitnexus/vendor/lbug-fts/prebuilds/darwin-arm64/libfts.lbug_extension b/gitnexus/vendor/lbug-fts/prebuilds/darwin-arm64/libfts.lbug_extension index 31f85a0da..53bcff048 100644 Binary files a/gitnexus/vendor/lbug-fts/prebuilds/darwin-arm64/libfts.lbug_extension and b/gitnexus/vendor/lbug-fts/prebuilds/darwin-arm64/libfts.lbug_extension differ diff --git a/gitnexus/vendor/lbug-fts/prebuilds/darwin-x64/libfts.lbug_extension b/gitnexus/vendor/lbug-fts/prebuilds/darwin-x64/libfts.lbug_extension index b0fd3b7fd..6fa05c0da 100644 Binary files a/gitnexus/vendor/lbug-fts/prebuilds/darwin-x64/libfts.lbug_extension and b/gitnexus/vendor/lbug-fts/prebuilds/darwin-x64/libfts.lbug_extension differ diff --git a/gitnexus/vendor/lbug-fts/prebuilds/linux-arm64/libfts.lbug_extension b/gitnexus/vendor/lbug-fts/prebuilds/linux-arm64/libfts.lbug_extension index 41f2fa575..bf4c01920 100644 Binary files a/gitnexus/vendor/lbug-fts/prebuilds/linux-arm64/libfts.lbug_extension and b/gitnexus/vendor/lbug-fts/prebuilds/linux-arm64/libfts.lbug_extension differ diff --git a/gitnexus/vendor/lbug-fts/prebuilds/linux-x64/libfts.lbug_extension b/gitnexus/vendor/lbug-fts/prebuilds/linux-x64/libfts.lbug_extension index 0971ff059..f7ab6c269 100644 Binary files a/gitnexus/vendor/lbug-fts/prebuilds/linux-x64/libfts.lbug_extension and b/gitnexus/vendor/lbug-fts/prebuilds/linux-x64/libfts.lbug_extension differ diff --git a/gitnexus/vendor/lbug-fts/prebuilds/win32-x64/libfts.lbug_extension b/gitnexus/vendor/lbug-fts/prebuilds/win32-x64/libfts.lbug_extension index 093f8b7f9..dec711432 100644 Binary files a/gitnexus/vendor/lbug-fts/prebuilds/win32-x64/libfts.lbug_extension and b/gitnexus/vendor/lbug-fts/prebuilds/win32-x64/libfts.lbug_extension differ diff --git a/gitnexus/vitest.config.ts b/gitnexus/vitest.config.ts index 29d44bf92..8bb6ff502 100644 --- a/gitnexus/vitest.config.ts +++ b/gitnexus/vitest.config.ts @@ -109,6 +109,7 @@ export default defineConfig({ 'test/integration/analyze-wal-checkpoint-failure.test.ts', 'test/integration/lbug-non-ascii-path.test.ts', 'test/integration/lbug-conn-serialization.test.ts', + 'test/integration/lbug-load-overlap-errors.test.ts', 'test/integration/load-cached-embeddings-spill.test.ts', 'test/integration/group/manifest-resolve-symbol-2325.test.ts', 'test/integration/group/manifest-synthetic-impact-lbug.test.ts', @@ -130,6 +131,8 @@ export default defineConfig({ // fake that answers on `query.includes(...)`. 'test/integration/wiki-graph-queries-engine.test.ts', 'test/unit/incremental-dirty-recovery.test.ts', + // Publication reconciliation uses real native COPY/checkpoints. + 'test/unit/incremental-write-integrity.test.ts', 'test/unit/incremental-orchestration.test.ts', // #2841. Native @ladybugdb/core: it runs real analyses, reopens the // DB under different extension-install policies, and reads @@ -190,6 +193,7 @@ export default defineConfig({ 'test/integration/analyze-wal-checkpoint-failure.test.ts', 'test/integration/lbug-non-ascii-path.test.ts', 'test/integration/lbug-conn-serialization.test.ts', + 'test/integration/lbug-load-overlap-errors.test.ts', 'test/integration/load-cached-embeddings-spill.test.ts', 'test/integration/group/manifest-resolve-symbol-2325.test.ts', 'test/integration/group/manifest-synthetic-impact-lbug.test.ts', @@ -206,6 +210,7 @@ export default defineConfig({ 'test/integration/detect-changes-path-anchoring.test.ts', 'test/integration/wiki-graph-queries-engine.test.ts', 'test/unit/incremental-dirty-recovery.test.ts', + 'test/unit/incremental-write-integrity.test.ts', 'test/unit/incremental-orchestration.test.ts', // Excluded here because it is included by `lbug-db` above; a file // in two projects would be collected (and run) twice.