GitNexus/eval/workflow_bench/review_cases/pr-2108.patch
Gergo Magyar ac7ae6a8ce Address PR review feedback (#2785)
Tighten review-evolution scoring, sandbox lock, and gateway cleanup so historical cells score instead of aborting or leaking host state.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-04 18:59:32 +00:00

845 lines
41 KiB
Diff
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

diff --git a/Dockerfile.cli b/Dockerfile.cli
index 3275c8f7e..cfba9bac4 100644
--- a/Dockerfile.cli
+++ b/Dockerfile.cli
@@ -67,6 +67,28 @@ COPY --from=builder --chown=node:node /app/gitnexus/vendor ./gitnexus/vendor
# unreachable from $PATH.
RUN ln -s /app/gitnexus/dist/cli/index.js /usr/local/bin/gitnexus
+# Bake the LadybugDB FTS extension into the image so BM25 keyword search works
+# at runtime. The server runs the default `load-only` extension policy (the read
+# pool pins `{ policy: 'load-only' }`), so a runtime `LOAD EXTENSION fts` never
+# INSTALLs — the extension must already exist in the runtime user's HOME
+# extension dir, or every keyword search silently degrades (no FTS indexes are
+# written and ranking falls back to vector-only with only a `warning` field).
+# Run the installer as the `node` user with the SAME HOME the server runs under,
+# so `INSTALL fts` materializes the extension under `$HOME/.lbdb/extension` where
+# the runtime `LOAD` resolves it offline. `ENV HOME` is pinned because Docker
+# does not derive HOME from `USER`, so without it build-install and runtime-load
+# would resolve different paths. Requires network egress for the one-time
+# INSTALL; the build fails loudly if it cannot fetch the extension. The DB-size
+# default comes from GITNEXUS_LBUG_MAX_DB_SIZE (single source of truth, matches
+# the runtime) — it only sizes the throwaway scratch DB used to run INSTALL.
+# The second `--verify-only` step re-LOADs the extension in a FRESH process
+# under the same HOME, so a HOME/extension-dir mismatch fails the build here
+# rather than silently degrading keyword search to vector-only at runtime.
+ENV HOME=/home/node \
+ GITNEXUS_LBUG_MAX_DB_SIZE=17179869184
+RUN su node -s /bin/sh -c "HOME=/home/node node /app/gitnexus/scripts/install-duckdb-extension.mjs fts" \
+ && su node -s /bin/sh -c "HOME=/home/node node /app/gitnexus/scripts/install-duckdb-extension.mjs fts --verify-only"
+
USER node
# The web UI defaults to http://localhost:4747 - keep that contract.
diff --git a/gitnexus/scripts/bench/fts-evict-reload-rss.mjs b/gitnexus/scripts/bench/fts-evict-reload-rss.mjs
new file mode 100644
index 000000000..6240e16b7
--- /dev/null
+++ b/gitnexus/scripts/bench/fts-evict-reload-rss.mjs
@@ -0,0 +1,421 @@
+#!/usr/bin/env node
+// FTS evict→reload RSS repro (gitnexus-enterprise PR #222 / local U3).
+//
+// Settles ONE empirical question that no static read can answer: when a
+// LadybugDB database that has `LOAD EXTENSION fts` applied is closed and a
+// fresh one is opened + re-LOADed (the pool's evict→reload cycle), does the
+// native FTS arena get reclaimed by `db.close()` — or is it stranded, so RSS
+// climbs without bound over a long-lived MCP `serve` session?
+//
+// • PLATEAU across cycles → db.close() reclaims the FTS arena; the OSS pool's
+// footprint is bounded by MAX_POOL_SIZE (~5 live arenas). No unbounded leak;
+// the #222 worker-isolation rewrite (plan U4) is NOT justified for OSS.
+// • MONOTONIC CLIMB → the FTS arena is stranded per reopen; the user's
+// hypothesis holds and U4 (route FTS reads through a reclaimable worker) is
+// justified.
+//
+// SCOPE OF THE VERDICT (read before citing it). A per-reload FTS-arena leak
+// would be PROPORTIONAL to the index size. A small fixture therefore produces a
+// small per-cycle increment that an absolute threshold can read as PLATEAU even
+// when a production-scale graph would leak visibly. So:
+// - `--rows` controls fixture size; run it LARGE (tens of thousands) before
+// concluding "no leak". The default is deliberately not tiny.
+// - The CLIMB gate combines a per-cycle slope with BOTH an absolute and a
+// per-row-relative delta floor, so the sensitivity scales with fixture size.
+// - The PLATEAU verdict is only valid for the corpus size it was run at; the
+// output states that size. The production-faithful confirmation is a
+// `--via-pool` run against a real large analyzed repo over a long session.
+//
+// Two modes:
+// (default) NATIVE — reproduces the native sequence doInitLbug()+closeOne()
+// perform (open Database → new Connection → LOAD EXTENSION fts →
+// QUERY_FTS_INDEX → close), against K self-built FTS fixtures, with no
+// gitnexus build required. `--no-await-close` mirrors the pool's
+// fire-and-forget close instead of awaiting (the production close shape).
+// --via-pool <lbugPath> — drives the REAL gitnexus pool from compiled dist
+// (initLbug → executeParameterized → closeLbug) against an existing analyzed
+// repo, exercising the production path + the GITNEXUS_POOL_RSS_TRACE
+// instrumentation. Probes ALL FTS indexes the repo has. Forces an explicit
+// close+reinit each cycle. Run `node scripts/build.js` first so the dist
+// reflects the current pool-adapter (incl. the RSS trace).
+//
+// Run with --expose-gc so RSS excludes V8-heap noise:
+// node --expose-gc gitnexus/scripts/bench/fts-evict-reload-rss.mjs
+// node --expose-gc gitnexus/scripts/bench/fts-evict-reload-rss.mjs --rows 40000 --cycles 30
+// GITNEXUS_POOL_RSS_TRACE=1 node --expose-gc \
+// gitnexus/scripts/bench/fts-evict-reload-rss.mjs --via-pool /path/to/repo/.gitnexus/lbug
+//
+// Flags by mode: --rows/--repos/--read-write/--no-await-close apply to NATIVE
+// only; --cycles applies to both. VIA-POOL warns when a NATIVE-only flag is set.
+//
+// Memory benches are noisy. Default is 24 cycles; trust the TREND (slope /
+// first-third vs last-third), never a single delta. A flat trend at a LARGE
+// fixture is a real NEGATIVE result (no unbounded leak), not a failed run.
+
+import { createRequire } from 'node:module';
+import os from 'node:os';
+import path from 'node:path';
+import fs from 'node:fs';
+
+const require = createRequire(import.meta.url);
+const lbugModule = require('@ladybugdb/core');
+const lbug = lbugModule.default ?? lbugModule;
+
+const LBUG_MAX_DB_SIZE = 16 * 1024 * 1024 * 1024;
+
+// ── args ──────────────────────────────────────────────────────────────────
+function argVal(flag, dflt) {
+ const i = process.argv.indexOf(flag);
+ return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : dflt;
+}
+const CYCLES = Math.max(6, parseInt(argVal('--cycles', '24'), 10) || 24);
+const REPOS = Math.max(1, parseInt(argVal('--repos', '6'), 10) || 6); // >5 mirrors LRU thrash
+// Fixture size. Default is large enough that a size-proportional leak would be
+// visible across cycles; raise it further before trusting a PLATEAU verdict.
+const ROWS = Math.max(100, parseInt(argVal('--rows', '8000'), 10) || 8000);
+const VIA_POOL = argVal('--via-pool', null);
+const READONLY = !process.argv.includes('--read-write');
+const AWAIT_CLOSE = !process.argv.includes('--no-await-close');
+
+if (VIA_POOL) {
+ // These flags are consumed only by NATIVE mode; warn rather than ignore
+ // silently so a VIA-POOL run is not misread as honoring them.
+ const ignored = ['--rows', '--repos', '--read-write', '--no-await-close'].filter((f) =>
+ process.argv.includes(f),
+ );
+ if (ignored.length) {
+ console.error(
+ `[fts-rss] NOTE: ${ignored.join(', ')} apply to NATIVE mode only; ignored in --via-pool.`,
+ );
+ }
+}
+
+if (typeof global.gc !== 'function') {
+ console.error(
+ '[fts-rss] WARNING: run with --expose-gc for clean RSS samples ' +
+ '(`node --expose-gc <thisfile>`). Continuing without forced GC — results are noisier.',
+ );
+}
+
+const gc = () => {
+ if (typeof global.gc === 'function') {
+ global.gc();
+ global.gc();
+ }
+};
+const rssMb = () => Math.round(process.memoryUsage().rss / (1024 * 1024));
+const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
+
+// ── fixture: a minimal FTS-bearing .lbug ────────────────────────────────────
+const WORDS = [
+ 'login auth session token user password validate verify credential',
+ 'parse tree syntax node grammar lexer token ast traversal visitor',
+ 'graph query cypher match relation node edge pattern aggregate index',
+ 'memory pool buffer arena allocate reclaim evict cache resident heap',
+ 'search rank score bm25 fts index stem porter keyword document corpus',
+ 'worker fork process spawn kill reclaim isolate native binding addon',
+];
+
+function buildFixture(dir) {
+ fs.mkdirSync(dir, { recursive: true });
+ const dbPath = path.join(dir, 'fixture.lbug');
+ const db = new lbug.Database(dbPath, 0, false, false, LBUG_MAX_DB_SIZE);
+ const conn = new lbug.Connection(db);
+ return (async () => {
+ await conn.query('LOAD EXTENSION fts');
+ await conn.query(
+ 'CREATE NODE TABLE Doc(id STRING, name STRING, content STRING, PRIMARY KEY(id))',
+ );
+ // Batch-insert via UNWIND so large fixtures (`--rows`) build in seconds
+ // instead of one round-trip per row. The fixture size drives the per-arena
+ // FTS allocation, which is what makes a size-proportional leak observable.
+ const rows = [];
+ for (let i = 0; i < ROWS; i++) {
+ const w = WORDS[i % WORDS.length];
+ const name = `sym_${i}`;
+ const content = `${w} ${name} block number ${i} ${WORDS[(i + 3) % WORDS.length]}`;
+ rows.push({ id: `doc:${i}`, name, content });
+ }
+ const INSERT_CHUNK = 2000;
+ for (let i = 0; i < rows.length; i += INSERT_CHUNK) {
+ const chunk = rows.slice(i, i + INSERT_CHUNK);
+ const stmt = await conn.prepare(
+ 'UNWIND $rows AS r CREATE (:Doc {id: r.id, name: r.name, content: r.content})',
+ );
+ await conn.execute(stmt, { rows: chunk });
+ }
+ await conn.query(
+ "CALL CREATE_FTS_INDEX('Doc', 'doc_fts', ['name', 'content'], stemmer := 'porter')",
+ );
+ await conn.close();
+ await db.close();
+ return dbPath;
+ })();
+}
+
+const QUERIES = ['login token', 'parse node', 'memory arena', 'search index', 'worker reclaim'];
+
+// ── NATIVE mode ─────────────────────────────────────────────────────────────
+async function runNative() {
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), 'fts-rss-'));
+ console.error(
+ `[fts-rss] NATIVE: ${REPOS} fixtures × ${ROWS} rows × ${CYCLES} cycles ` +
+ `(readOnly=${READONLY}, awaitClose=${AWAIT_CLOSE})`,
+ );
+ console.error(`[fts-rss] building ${REPOS} FTS fixture(s) under ${root} …`);
+
+ const srcDb = await buildFixture(path.join(root, 'src'));
+ const repoPaths = [];
+ for (let k = 0; k < REPOS; k++) {
+ const dst = path.join(root, `repo-${k}`);
+ fs.cpSync(path.dirname(srcDb), dst, { recursive: true });
+ repoPaths.push(path.join(dst, 'fixture.lbug'));
+ }
+
+ // Mirror the pool's evict→reload: each visit opens a FRESH Database, makes a
+ // Connection, LOADs fts, runs an FTS query, then closes — no caching, so every
+ // visit is a reload. K>5 amplifies the LRU-thrash signal the pool would see.
+ const series = [];
+ gc();
+ await sleep(50);
+ const baseline = rssMb();
+ console.error(`[fts-rss] baseline RSS=${baseline}MB`);
+
+ for (let cycle = 0; cycle < CYCLES; cycle++) {
+ for (let k = 0; k < REPOS; k++) {
+ const db = new lbug.Database(repoPaths[k], 0, false, READONLY, LBUG_MAX_DB_SIZE);
+ const conn = new lbug.Connection(db);
+ try {
+ await conn.query('LOAD EXTENSION fts'); // the per-reload re-LOAD under test
+ const q = QUERIES[(cycle + k) % QUERIES.length];
+ const res = await conn.query(
+ `CALL QUERY_FTS_INDEX('Doc', 'doc_fts', '${q}') RETURN node.id AS id, score ORDER BY score DESC LIMIT 20`,
+ );
+ // Drain so the query actually materializes results.
+ if (res && typeof res.getAll === 'function') await res.getAll();
+ } catch (e) {
+ console.error(`[fts-rss] query error (cycle ${cycle}, repo ${k}): ${e?.message || e}`);
+ } finally {
+ // AWAIT_CLOSE (default) is the best case for reclamation. --no-await-close
+ // mirrors the pool's fire-and-forget close (closeOne: db.close().catch())
+ // so a leak that only manifests without awaiting is not hidden.
+ if (AWAIT_CLOSE) {
+ try {
+ await conn.close();
+ await db.close();
+ } catch {
+ /* ignore */
+ }
+ } else {
+ conn.close().catch(() => {});
+ db.close().catch(() => {});
+ }
+ }
+ }
+ gc();
+ // Longer settle when not awaiting close, so fire-and-forget native teardown
+ // has a chance to complete before the RSS sample (avoids a false PLATEAU).
+ await sleep(AWAIT_CLOSE ? 20 : 200);
+ const rss = rssMb();
+ series.push(rss);
+ console.error(`[fts-rss] cycle ${String(cycle + 1).padStart(3)}/${CYCLES} rssMB=${rss}`);
+ }
+
+ fs.rmSync(root, { recursive: true, force: true });
+ return { baseline, series, corpus: `${REPOS}×${ROWS} rows, native, awaitClose=${AWAIT_CLOSE}` };
+}
+
+// ── VIA-POOL mode (real gitnexus pool from compiled dist) ───────────────────
+async function runViaPool(lbugPath) {
+ if (!fs.existsSync(lbugPath)) {
+ console.error(`[fts-rss] --via-pool path not found: ${lbugPath}`);
+ process.exit(2);
+ }
+ // Compiled dist is required (the pool pulls the native addon + many modules).
+ const distUrl = new URL('../../dist/core/lbug/pool-adapter.js', import.meta.url);
+ let pool;
+ try {
+ pool = await import(distUrl.href);
+ } catch (e) {
+ console.error(
+ `[fts-rss] could not import compiled pool-adapter (${e?.message}). ` +
+ `Run \`node scripts/build.js\` first, or use NATIVE mode.`,
+ );
+ process.exit(2);
+ }
+ const { initLbug, executeParameterized, closeLbug } = pool;
+ console.error(
+ `[fts-rss] VIA-POOL on ${lbugPath} × ${CYCLES} cycles ` +
+ `(explicit closeLbug+initLbug per cycle = forced evict→reload)`,
+ );
+
+ // Probe ALL FTS indexes the analyzed graph carries (mirrors fts-schema.ts
+ // FTS_INDEXES) so the per-cycle FTS arena load matches production, not a
+ // 2-of-5 subset that would understate it.
+ const FTS_INDEXES = [
+ { table: 'File', indexName: 'file_fts' },
+ { table: 'Function', indexName: 'function_fts' },
+ { table: 'Class', indexName: 'class_fts' },
+ { table: 'Method', indexName: 'method_fts' },
+ { table: 'Interface', indexName: 'interface_fts' },
+ ];
+
+ const series = [];
+ gc();
+ const baseline = rssMb();
+ console.error(`[fts-rss] baseline RSS=${baseline}MB`);
+
+ for (let cycle = 0; cycle < CYCLES; cycle++) {
+ try {
+ await initLbug(lbugPath, lbugPath);
+ const q = QUERIES[cycle % QUERIES.length];
+ for (const { table, indexName } of FTS_INDEXES) {
+ await executeParameterized(
+ lbugPath,
+ `CALL QUERY_FTS_INDEX('${table}', '${indexName}', $q) RETURN node.id AS id, score ORDER BY score DESC LIMIT 20`,
+ { q },
+ ).catch(() => []); // index may not exist for this graph — that's fine
+ }
+ await closeLbug(lbugPath); // force eviction → next cycle reopens + re-LOADs fts
+ } catch (e) {
+ console.error(`[fts-rss] pool cycle ${cycle} error: ${e?.message || e}`);
+ }
+ gc();
+ // closeLbug fires a fire-and-forget native close (pool closeOne:
+ // db.close().catch()), so settle longer than NATIVE's awaited close to let
+ // native teardown finish before sampling — else a real leak reads PLATEAU.
+ await sleep(200);
+ const rss = rssMb();
+ series.push(rss);
+ console.error(`[fts-rss] cycle ${String(cycle + 1).padStart(3)}/${CYCLES} rssMB=${rss}`);
+ }
+ await closeLbug().catch(() => {});
+ return { baseline, series, corpus: `via-pool ${path.basename(path.dirname(lbugPath))}` };
+}
+
+// ── verdict ─────────────────────────────────────────────────────────────────
+function median(xs) {
+ const s = [...xs].sort((a, b) => a - b);
+ const m = Math.floor(s.length / 2);
+ return s.length % 2 ? s[m] : Math.round((s[m - 1] + s[m]) / 2);
+}
+function slopeMbPerCycle(series) {
+ // Least-squares slope of rss vs cycle index.
+ const n = series.length;
+ const xs = series.map((_, i) => i);
+ const xMean = xs.reduce((a, b) => a + b, 0) / n;
+ const yMean = series.reduce((a, b) => a + b, 0) / n;
+ let num = 0,
+ den = 0;
+ for (let i = 0; i < n; i++) {
+ num += (xs[i] - xMean) * (series[i] - yMean);
+ den += (xs[i] - xMean) ** 2;
+ }
+ return den === 0 ? 0 : num / den;
+}
+
+function verdict({ baseline, series, corpus }) {
+ const third = Math.max(1, Math.floor(series.length / 3));
+ const firstMed = median(series.slice(0, third));
+ const lastMed = median(series.slice(-third));
+ const delta = lastMed - firstMed;
+ const slope = slopeMbPerCycle(series);
+ const peak = Math.max(...series);
+
+ // The discriminant between a real leak and allocator warmup is SLOPE
+ // DECELERATION, not total delta. Both a leak and a warmup-to-plateau climb;
+ // they differ in whether the per-cycle increment is SUSTAINED or DECAYS:
+ // - true per-reload leak (stranded FTS arena): RSS rises ~linearly, so the
+ // second-half slope ≈ the first-half slope (the increment does not decay).
+ // - allocator working-set warmup (native free-pool growing to the working
+ // set, freed pages retained-then-reused): RSS rises then flattens, so the
+ // second-half slope is a small FRACTION of the first-half slope. Larger
+ // fixtures warm up over MORE cycles, which a fixed absolute-delta gate
+ // misreads as a leak — the slope ratio is scale-invariant and does not.
+ const half = Math.max(1, Math.floor(series.length / 2));
+ const firstHalfSlope = slopeMbPerCycle(series.slice(0, half));
+ const secondHalfSlope = slopeMbPerCycle(series.slice(-half));
+
+ // Detect a STEP DISCONTINUITY — a single cycle-to-cycle jump far larger than
+ // the typical per-cycle delta. A one-time allocator/arena reservation jump
+ // (then flat) is NOT a per-reload leak, but it inflates the second-half slope
+ // and would fool a pure slope test; it also signals a noisy run.
+ const deltas = series.slice(1).map((v, i) => v - series[i]);
+ const absDeltas = deltas.map(Math.abs).sort((a, b) => a - b);
+ const medAbsDelta = absDeltas.length ? absDeltas[Math.floor(absDeltas.length / 2)] : 0;
+ const maxJump = deltas.length ? Math.max(...deltas) : 0;
+ const stepDiscontinuity = maxJump > Math.max(30, 5 * Math.max(medAbsDelta, 1));
+
+ // The discriminant between a real leak and allocator warmup is SLOPE
+ // DECELERATION, not total delta. A true per-reload leak (stranded FTS arena)
+ // rises ~linearly: the second-half slope stays ≈ the first-half slope. An
+ // allocator working-set warmup rises then flattens: the second-half slope is
+ // a small FRACTION of the first-half. Larger fixtures warm up over MORE
+ // cycles, which a fixed absolute-delta gate misreads as a leak — the slope
+ // ratio is scale-invariant. Below SUSTAIN_FLOOR (~0.5 MB/cycle) the tail is
+ // effectively flat (noise).
+ const SUSTAIN_FLOOR = 0.5;
+ const decelRatio = secondHalfSlope / Math.max(firstHalfSlope, 1e-9);
+ let label;
+ if (stepDiscontinuity) {
+ // A discrete jump (then flat) is not linear accumulation, but the run is
+ // noisy — don't claim a clean result either way.
+ label = 'INCONCLUSIVE';
+ } else if (secondHalfSlope < SUSTAIN_FLOOR) {
+ label = 'PLATEAU';
+ } else if (decelRatio >= 0.6) {
+ label = 'CLIMB';
+ } else {
+ // Tail slope above the flat floor but clearly decelerating — converging,
+ // but not yet flat. Honest answer at this corpus is "not resolved".
+ label = 'INCONCLUSIVE';
+ }
+
+ console.log('\n==================== FTS evict→reload RSS verdict ====================');
+ console.log(`corpus: ${corpus}`);
+ console.log(`samples (MB): ${series.join(' ')}`);
+ console.log(
+ `baseline=${baseline} firstThirdMed=${firstMed} lastThirdMed=${lastMed} delta=${delta}MB ` +
+ `peak=${peak} overallSlope=${slope.toFixed(2)} firstHalfSlope=${firstHalfSlope.toFixed(2)} ` +
+ `secondHalfSlope=${secondHalfSlope.toFixed(2)}MB/cycle maxJump=${maxJump}MB step=${stepDiscontinuity} cycles=${series.length}`,
+ );
+ if (label === 'CLIMB') {
+ console.log(
+ 'VERDICT: CLIMB — the per-cycle increment is SUSTAINED (second-half slope ≈ first-half),\n' +
+ ' i.e. RSS rises ~linearly with no decay. The native FTS arena is NOT reclaimed\n' +
+ ' by db.close(); the leak is real over a long-lived session.\n' +
+ ' → plan U4 (worker/process isolation of the FTS read path) is JUSTIFIED.',
+ );
+ } else if (label === 'PLATEAU') {
+ console.log(
+ `VERDICT: PLATEAU at this corpus (${corpus}) — the per-cycle increment DECAYS to flat\n` +
+ ' (second-half slope below the noise floor). db.close() reclaims the FTS arena;\n' +
+ ' footprint is bounded (and the pool further caps it at MAX_POOL_SIZE). No\n' +
+ ' unbounded leak. Caveat: synthetic fixture — confirm with a --via-pool run\n' +
+ ' against a real large analyzed repo before fully closing plan U4.',
+ );
+ } else {
+ console.log(
+ `VERDICT: INCONCLUSIVE at this corpus (${corpus}) — the run is noisy (step discontinuity)\n` +
+ ' or still decelerating without reaching flat, so neither a clean PLATEAU nor a\n' +
+ ' sustained linear CLIMB can be asserted. NATIVE synthetic runs do not resolve\n' +
+ ' this reliably at scale. The definitive test is a --via-pool run against a real\n' +
+ ' large analyzed repo over many cycles (with GITNEXUS_POOL_RSS_TRACE=1). Plan U4\n' +
+ ' stays GATED — neither closed nor built on this evidence.',
+ );
+ }
+ console.log(
+ `MACHINE: ${JSON.stringify({ mode: VIA_POOL ? 'via-pool' : 'native', corpus, baseline, firstMed, lastMed, delta, overallSlope: Number(slope.toFixed(3)), firstHalfSlope: Number(firstHalfSlope.toFixed(3)), secondHalfSlope: Number(secondHalfSlope.toFixed(3)), maxJump, stepDiscontinuity, peak, cycles: series.length, verdict: label })}`,
+ );
+ console.log('=====================================================================\n');
+}
+
+// ── main ────────────────────────────────────────────────────────────────────
+(async () => {
+ const result = VIA_POOL ? await runViaPool(VIA_POOL) : await runNative();
+ verdict(result);
+ process.exit(0);
+})().catch((e) => {
+ console.error('[fts-rss] fatal:', e?.stack || e);
+ process.exit(1);
+});
diff --git a/gitnexus/scripts/install-duckdb-extension.mjs b/gitnexus/scripts/install-duckdb-extension.mjs
index 2bc65a05e..7492e084f 100644
--- a/gitnexus/scripts/install-duckdb-extension.mjs
+++ b/gitnexus/scripts/install-duckdb-extension.mjs
@@ -14,7 +14,7 @@ function parseLbugMaxDbSize(raw) {
return Math.floor(parsed);
}
-async function installDuckDbExtension(extensionName) {
+async function installDuckDbExtension(extensionName, verifyOnly = false) {
if (!extensionName || !EXTENSION_NAME_PATTERN.test(extensionName)) {
throw new Error(`Invalid DuckDB extension name: ${extensionName ?? '<missing>'}`);
}
@@ -22,9 +22,11 @@ async function installDuckDbExtension(extensionName) {
const require = createRequire(import.meta.url);
const lbugModule = require('@ladybugdb/core');
const lbug = lbugModule.default ?? lbugModule;
- const lbugMaxDbSize = parseLbugMaxDbSize(
- process.argv[3] ?? process.env.GITNEXUS_LBUG_MAX_DB_SIZE,
- );
+ // argv[3] is the optional positional size; ignore it when it is actually a
+ // flag token (e.g. `--verify-only`) and fall back to the env default.
+ const sizeArg =
+ process.argv[3] && !process.argv[3].startsWith('--') ? process.argv[3] : undefined;
+ const lbugMaxDbSize = parseLbugMaxDbSize(sizeArg ?? process.env.GITNEXUS_LBUG_MAX_DB_SIZE);
const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-ext-install-'));
const dbPath = path.join(tmpDir, 'install.lbug');
@@ -34,7 +36,18 @@ async function installDuckDbExtension(extensionName) {
try {
db = new lbug.Database(dbPath, 0, false, false, lbugMaxDbSize);
conn = new lbug.Connection(db);
- await conn.query(`INSTALL ${extensionName}`);
+ if (verifyOnly) {
+ // Prove a previously-baked extension is resolvable by a FRESH process
+ // under the current HOME (the runtime `LOAD EXTENSION` path) — no INSTALL,
+ // no network. Used as a Docker build-time gate so a HOME/extension-dir
+ // mismatch fails the build instead of silently degrading search at runtime.
+ await conn.query(`LOAD EXTENSION ${extensionName}`);
+ console.log(
+ `[install-ext] LOAD-only verify OK for '${extensionName}' (HOME=${process.env.HOME})`,
+ );
+ } else {
+ await conn.query(`INSTALL ${extensionName}`);
+ }
} finally {
if (conn) await conn.close().catch(() => {});
if (db) await db.close().catch(() => {});
@@ -42,7 +55,10 @@ async function installDuckDbExtension(extensionName) {
}
}
-installDuckDbExtension(process.argv[2] ?? process.env.GITNEXUS_LBUG_EXTENSION_NAME).catch((err) => {
+installDuckDbExtension(
+ process.argv[2] ?? process.env.GITNEXUS_LBUG_EXTENSION_NAME,
+ process.argv.includes('--verify-only'),
+).catch((err) => {
console.error(err instanceof Error ? (err.stack ?? err.message) : String(err));
process.exitCode = 1;
});
diff --git a/gitnexus/src/core/lbug/pool-adapter.ts b/gitnexus/src/core/lbug/pool-adapter.ts
index 030688d14..e5338e3d0 100644
--- a/gitnexus/src/core/lbug/pool-adapter.ts
+++ b/gitnexus/src/core/lbug/pool-adapter.ts
@@ -103,6 +103,19 @@ const IDLE_TIMEOUT_MS = 5 * 60 * 1000; // 5 minutes
/** Max connections per repo (caps concurrent queries per repo) */
const MAX_CONNS_PER_REPO = 8;
+// Behavior-neutral RSS tracing for the FTS evict→reload memory repro
+// (gitnexus/scripts/bench/fts-evict-reload-rss.mjs). Two invariants keep it safe
+// in the pool init/close hot path: it writes ONLY to stderr (stdout is the MCP
+// JSON-RPC channel), and the GITNEXUS_POOL_RSS_TRACE gate makes it a no-op — one
+// env-var compare per call, nothing else — unless a harness explicitly enables it.
+function traceRss(event: 'init' | 'close', repoId: string): void {
+ if (process.env.GITNEXUS_POOL_RSS_TRACE !== '1') return;
+ const rssMb = Math.round(process.memoryUsage().rss / (1024 * 1024));
+ process.stderr.write(
+ `[pool-rss] ${event} repo=${repoId} pool=${pool.size} dbCache=${dbCache.size} rssMB=${rssMb}\n`,
+ );
+}
+
let idleTimer: ReturnType<typeof setInterval> | null = null;
// Stdout-capture state lives in `gitnexus/src/mcp/stdio-capture.ts` — a leaf
@@ -240,6 +253,8 @@ function closeOne(repoId: string): void {
// Isolate listener failures — teardown must complete.
}
}
+
+ traceRss('close', repoId);
}
/**
@@ -611,6 +626,7 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
closed: false,
});
ensureIdleTimer();
+ traceRss('init', repoId);
}
/**
@@ -673,6 +689,7 @@ export async function initLbugWithDb(
closed: false,
});
ensureIdleTimer();
+ traceRss('init', repoId);
}
/**
diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts
index 77411b107..c0a30eb58 100644
--- a/gitnexus/src/mcp/local/local-backend.ts
+++ b/gitnexus/src/mcp/local/local-backend.ts
@@ -1112,73 +1112,120 @@ export class LocalBackend {
>();
const definitions: any[] = []; // standalone symbols not in any process
- for (const [_, item] of merged) {
- const sym = item.data;
- if (!sym.nodeId) {
- // File-level results go to definitions
- definitions.push({
- name: sym.name,
- type: sym.type || 'File',
- filePath: sym.filePath,
- });
- continue;
- }
-
- // Find processes this symbol participates in
- let processRows: any[] = [];
+ // Batch-fetch process participation, cohesion, and (optionally) content for
+ // ALL matched symbols in 2-3 graph queries instead of 2-3 *per symbol*. The
+ // previous per-symbol loop issued up to 3N sequential pool round-trips
+ // (searchLimit symbols × {STEP_IN_PROCESS, MEMBER_OF, content}); on a warm
+ // repo the IPC + query-setup overhead of those round-trips dominated query
+ // latency. Collapsing to `WHERE n.id IN $nodeIds` preserves identical output
+ // (the aggregation loop below is unchanged) while cutting the round-trips.
+ // Array params bind through the pool exactly as bm25Search's
+ // `WHERE n.id IN $nodeIds` already does. (Ported from gitnexus-enterprise
+ // PR #222 — N+1 → 2-3 batched queries.)
+ const nodeIds = merged.map(([, m]) => m.data?.nodeId).filter((id): id is string => !!id);
+
+ const processRowsByNode = new Map<string, any[]>();
+ const cohesionByNode = new Map<string, { cohesion: number; module?: string }>();
+ const contentByNode = new Map<string, string>();
+
+ // Chunk the IN-list like the impact path (CHUNK_SIZE=100) so a large result
+ // set never builds an unbounded `IN` parameter. Default batch is
+ // processLimit*maxSymbolsPerProcess (≤ one chunk), but chunk for robustness.
+ const QUERY_CHUNK_SIZE = 100;
+ for (let i = 0; i < nodeIds.length; i += QUERY_CHUNK_SIZE) {
+ const ids = nodeIds.slice(i, i + QUERY_CHUNK_SIZE);
+
+ // Processes each symbol participates in. `n.id AS nodeId` is prepended as
+ // column 0 so rows from many symbols can be re-associated to their symbol.
try {
- processRows = await executeParameterized(
+ const rows = await executeParameterized(
repo.lbugPath,
`
- MATCH (n {id: $nodeId})-[r:CodeRelation {type: 'STEP_IN_PROCESS'}]->(p:Process)
- RETURN p.id AS pid, p.label AS label, p.heuristicLabel AS heuristicLabel, p.processType AS processType, p.stepCount AS stepCount, r.step AS step
+ MATCH (n)-[r:CodeRelation {type: 'STEP_IN_PROCESS'}]->(p:Process)
+ WHERE n.id IN $nodeIds
+ RETURN n.id AS nodeId, p.id AS pid, p.label AS label, p.heuristicLabel AS heuristicLabel, p.processType AS processType, p.stepCount AS stepCount, r.step AS step
`,
- { nodeId: sym.nodeId },
+ { nodeIds: ids },
);
+ for (const row of rows) {
+ const nid = row.nodeId ?? row[0];
+ let list = processRowsByNode.get(nid);
+ if (!list) processRowsByNode.set(nid, (list = []));
+ list.push(row);
+ }
} catch (e) {
logQueryError('query:process-lookup', e);
}
- // Get cluster membership + cohesion (cohesion used as internal ranking signal)
- let cohesion = 0;
- let module: string | undefined;
+ // Cluster membership + cohesion. Keep the FIRST community row per node to
+ // mirror the prior per-symbol `LIMIT 1` (each symbol keeps ITS community,
+ // not one community for the whole batch).
try {
- const cohesionRows = await executeParameterized(
+ const rows = await executeParameterized(
repo.lbugPath,
`
- MATCH (n {id: $nodeId})-[:CodeRelation {type: 'MEMBER_OF'}]->(c:Community)
- RETURN c.cohesion AS cohesion, c.heuristicLabel AS module
- LIMIT 1
+ MATCH (n)-[:CodeRelation {type: 'MEMBER_OF'}]->(c:Community)
+ WHERE n.id IN $nodeIds
+ RETURN n.id AS nodeId, c.cohesion AS cohesion, c.heuristicLabel AS module
`,
- { nodeId: sym.nodeId },
+ { nodeIds: ids },
);
- if (cohesionRows.length > 0) {
- cohesion = (cohesionRows[0].cohesion ?? cohesionRows[0][0]) || 0;
- module = cohesionRows[0].module ?? cohesionRows[0][1];
+ for (const row of rows) {
+ const nid = row.nodeId ?? row[0];
+ if (!cohesionByNode.has(nid)) {
+ cohesionByNode.set(nid, {
+ cohesion: (row.cohesion ?? row[1]) || 0,
+ module: row.module ?? row[2],
+ });
+ }
}
} catch (e) {
logQueryError('query:cluster-info', e);
}
- // Optionally fetch content
- let content: string | undefined;
+ // Optionally fetch content for every matched symbol.
if (includeContent) {
try {
- const contentRows = await executeParameterized(
+ const rows = await executeParameterized(
repo.lbugPath,
`
- MATCH (n {id: $nodeId})
- RETURN n.content AS content
+ MATCH (n)
+ WHERE n.id IN $nodeIds
+ RETURN n.id AS nodeId, n.content AS content
`,
- { nodeId: sym.nodeId },
+ { nodeIds: ids },
);
- if (contentRows.length > 0) {
- content = contentRows[0].content ?? contentRows[0][0];
+ for (const row of rows) {
+ const nid = row.nodeId ?? row[0];
+ contentByNode.set(nid, row.content ?? row[1]);
}
} catch (e) {
logQueryError('query:content-fetch', e);
}
}
+ }
+
+ // Aggregation is unchanged from the per-symbol version — it now reads the
+ // pre-fetched maps instead of issuing a query per symbol. Iterating `merged`
+ // in the same (sorted) order preserves processMap insertion order, the
+ // definitions order, and the item.score association exactly.
+ for (const [_, item] of merged) {
+ const sym = item.data;
+ if (!sym.nodeId) {
+ // File-level results go to definitions
+ definitions.push({
+ name: sym.name,
+ type: sym.type || 'File',
+ filePath: sym.filePath,
+ });
+ continue;
+ }
+
+ const processRows = processRowsByNode.get(sym.nodeId) ?? [];
+ const coh = cohesionByNode.get(sym.nodeId);
+ const cohesion = coh?.cohesion ?? 0;
+ const module = coh?.module;
+ const content = includeContent ? contentByNode.get(sym.nodeId) : undefined;
const symbolEntry = {
id: sym.nodeId,
@@ -1197,12 +1244,13 @@ export class LocalBackend {
} else {
// Add to each process it belongs to
for (const row of processRows) {
- const pid = row.pid ?? row[0];
- const label = row.label ?? row[1];
- const hLabel = row.heuristicLabel ?? row[2];
- const pType = row.processType ?? row[3];
- const stepCount = row.stepCount ?? row[4];
- const step = row.step ?? row[5];
+ // Positional fallbacks shift +1 because `n.id AS nodeId` is column 0.
+ const pid = row.pid ?? row[1];
+ const label = row.label ?? row[2];
+ const hLabel = row.heuristicLabel ?? row[3];
+ const pType = row.processType ?? row[4];
+ const stepCount = row.stepCount ?? row[5];
+ const step = row.step ?? row[6];
if (!processMap.has(pid)) {
processMap.set(pid, {
diff --git a/gitnexus/test/fixtures/local-backend-seed.ts b/gitnexus/test/fixtures/local-backend-seed.ts
index 3f299046e..4878348b2 100644
--- a/gitnexus/test/fixtures/local-backend-seed.ts
+++ b/gitnexus/test/fixtures/local-backend-seed.ts
@@ -35,6 +35,12 @@ export const LOCAL_BACKEND_SEED_DATA = [
CREATE (a)-[:CodeRelation {type: 'STEP_IN_PROCESS', confidence: 1.0, reason: '', step: 1}]->(p)`,
`MATCH (a:Function), (p:Process) WHERE a.id = 'func:validate' AND p.id = 'proc:login-flow'
CREATE (a)-[:CodeRelation {type: 'STEP_IN_PROCESS', confidence: 1.0, reason: '', step: 2}]->(p)`,
+ // func:validate is the terminalId of proc:beta-flow too — wiring its second
+ // STEP_IN_PROCESS edge makes it a genuine MULTI-process symbol, which the
+ // batched-query test uses to exercise the full row[1..6] positional shift
+ // (a single-process symbol can't expose an off-by-one in those fallbacks).
+ `MATCH (a:Function), (p:Process) WHERE a.id = 'func:validate' AND p.id = 'proc:beta-flow'
+ CREATE (a)-[:CodeRelation {type: 'STEP_IN_PROCESS', confidence: 1.0, reason: '', step: 3}]->(p)`,
`MATCH (h:Function), (t:Tool) WHERE h.id = 'func:alpha' AND t.id = 'Tool:alpha'
CREATE (h)-[:CodeRelation {type: 'HANDLES_TOOL', confidence: 1.0, reason: 'tool-definition', step: 0}]->(t)`,
`MATCH (h:Function), (t:Tool) WHERE h.id = 'func:beta' AND t.id = 'Tool:beta'
diff --git a/gitnexus/test/integration/local-backend-calltool.test.ts b/gitnexus/test/integration/local-backend-calltool.test.ts
index e2640deab..176e061b2 100644
--- a/gitnexus/test/integration/local-backend-calltool.test.ts
+++ b/gitnexus/test/integration/local-backend-calltool.test.ts
@@ -113,6 +113,72 @@ withTestLbugDB(
expect(result.timing.bm25 ?? result.timing.vector).toBeGreaterThanOrEqual(0);
});
+ // PR #222 port: the query tool batches per-symbol process/cohesion/content
+ // lookups (N+1 → 2-3 `WHERE n.id IN $nodeIds` queries). These assertions
+ // guard the batch-adaptation hazards that a naive cherry-pick would break:
+ // (1) each symbol keeps ITS OWN community (the per-node first-row pick that
+ // replaced the per-symbol `LIMIT 1`), and (2) content maps to the right
+ // node — both depend on the +1 positional-index shift after prepending
+ // `n.id AS nodeId`. func:login is MEMBER_OF comm:auth ("Authentication");
+ // func:validate has no community, so it must NOT inherit login's.
+ it('query batches per-symbol enrichment without cross-assigning community/content', async () => {
+ const findSym = (res: any, id: string) =>
+ (res.process_symbols ?? []).find((s: any) => s.id === id) ??
+ (res.definitions ?? []).find((s: any) => s.id === id);
+
+ const loginRes = await backend.callTool('query', {
+ query: 'login',
+ include_content: true,
+ });
+ expect(loginRes).not.toHaveProperty('error');
+ const login = findSym(loginRes, 'func:login');
+ expect(login).toBeDefined();
+ // Community correctly associated to its own node (not dropped, not leaked).
+ expect(login.module).toBe('Authentication');
+ // Content correctly mapped to its own node (positional [1] after nodeId).
+ expect(login.content).toBe('function login() {}');
+
+ const validateRes = await backend.callTool('query', {
+ query: 'validate',
+ include_content: true,
+ });
+ expect(validateRes).not.toHaveProperty('error');
+ const validate = findSym(validateRes, 'func:validate');
+ expect(validate).toBeDefined();
+ // validate has no MEMBER_OF edge — a flat batched `LIMIT 1` would have
+ // leaked some other node's community onto it. It must have none.
+ expect(validate.module).toBeUndefined();
+ expect(validate.content).toBe('function validate() {}');
+ });
+
+ // PR #222 port: a symbol in MULTIPLE processes is what fully exercises the
+ // +1 positional shift in the batched STEP_IN_PROCESS aggregation — with a
+ // single process row, `row.pid ?? row[1]` succeeds whether the shift is
+ // right or wrong. func:validate is a step in BOTH proc:login-flow (step 2)
+ // and proc:beta-flow (step 3), so both rows for the one node must be parsed
+ // (pid=row[1], step=row[6]); an off-by-one would drop a process or mis-pair
+ // pid↔step. Also pins process ranking (totalScore via the regroup-by-nodeId).
+ it('query batches a multi-process symbol and ranks processes (positional shift across rows)', async () => {
+ const res = await backend.callTool('query', { query: 'validate' });
+ expect(res).not.toHaveProperty('error');
+ const processIds = (res.processes ?? []).map((p: any) => p.id);
+ // Both of validate's processes must appear — both STEP_IN_PROCESS rows
+ // were parsed and grouped by the correct pid (row[1]).
+ expect(processIds).toContain('proc:login-flow');
+ expect(processIds).toContain('proc:beta-flow');
+
+ // process_symbols dedups by id, so validate appears once carrying the
+ // pid+step of its top-ranked process — they must come from the SAME
+ // shifted row: login-flow⇒step 2, beta-flow⇒step 3.
+ const v = (res.process_symbols ?? []).find((s: any) => s.id === 'func:validate');
+ expect(v).toBeDefined();
+ expect(v.step_index).toBe(v.process_id === 'proc:beta-flow' ? 3 : 2);
+
+ // Ranking: 'login' surfaces proc:login-flow as the top process.
+ const loginRes = await backend.callTool('query', { query: 'login' });
+ expect((loginRes.processes ?? [])[0]?.id).toBe('proc:login-flow');
+ });
+
it('tool_map returns per-tool flows without cross-attributing same-file tools', async () => {
const result = await backend.callTool('tool_map', {});
expect(result).not.toHaveProperty('error');