From ad5ff42804d9ec15a7b3d5cc9ab6648f013f435e Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 15:45:31 +0000 Subject: [PATCH 01/10] feat(lbug): add adaptive buffer-pool size hint Adds the sizing lever without changing behavior yet: a module-scoped buffer-pool size hint plus estimateBufferPool(graphElementCount), read by resolveBufferManagerSize with precedence env-override > clamp(hint, 64MiB, default) > default. The hint can only shrink the pool from the default (clamped to [floor, default]), so the 2GiB/80%-RAM cap and the GITNEXUS_LBUG_BUFFER_POOL_SIZE escape hatch (incl. 0) are preserved. With no hint set, resolveBufferManagerSize returns exactly what it did before. Motivation: LadybugDB eagerly commits the buffer pool at DB open, so the fixed min(2GiB,80%RAM) pool adds a measured ~2.8s to every analyze even on a 3-file repo (dominant on Windows). Sizing the pool to the graph lets small repos use the fast 64MiB floor while large repos keep the cap. Co-Authored-By: Claude Fable 5 --- gitnexus/src/core/lbug/lbug-config.ts | 54 ++++++++++++++-- gitnexus/test/unit/lbug-config-wal.test.ts | 71 +++++++++++++++++++++- 2 files changed, 120 insertions(+), 5 deletions(-) diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index 540fe530a..f32a12898 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -333,16 +333,62 @@ const parseBufferPoolSize = (raw: string | undefined): number | undefined => { const defaultBufferPoolSize = (): number => Math.min(DEFAULT_BUFFER_POOL_CAP, Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8))); +/** Clamp a requested pool size to [floor, default] — never above the 2 GiB / 80%-RAM default. */ +const clampBufferPool = (bytes: number): number => + Math.min(defaultBufferPoolSize(), Math.max(BUFFER_POOL_FLOOR, Math.floor(bytes))); + +/** + * Buffer-pool bytes to provision per graph element (node + relationship). + * + * LadybugDB eagerly commits the buffer pool at DB open, so an oversized pool + * costs a fixed penalty on every analyze regardless of repo size (measured: + * ~2.8 s extra for a pool ≥128 MiB vs the 64 MiB floor, on a 3-file repo). + * The pool is only a page cache over the on-disk index, and the index scales + * with node/edge count — so a per-element budget lets a small repo run with a + * small, fast-to-commit pool while a large repo still reaches the 2 GiB cap. + * Kept deliberately generous (holds the working set without thrash); tuned by + * `bench/buffer-pool/measure.mjs` against a large-repo analyze. + */ +const POOL_BYTES_PER_ELEMENT = 4 * 1024; + +/** + * Size the buffer pool to an estimated graph size (node + relationship count), + * clamped to [BUFFER_POOL_FLOOR, defaultBufferPoolSize()]. The estimate can + * only *shrink* the pool from the default — it never exceeds the 2 GiB / 80%-RAM + * cap — so a large repo is never under the default it gets today. + */ +export const estimateBufferPool = (graphElementCount: number): number => + clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT); + +/** + * Optional per-run buffer-pool size hint (bytes). The analyze orchestrator sets + * it once the graph size is known (after the pipeline, before the DB open) so a + * small repo opens with a small pool, and clears it at run end. Non-analyze + * opens (MCP serve, `native-check` `:memory:`) never set it and keep the default. + */ +let bufferPoolSizeHint: number | undefined; + +/** Set (bytes) or clear (`undefined`) the per-run buffer-pool size hint. */ +export const setBufferPoolSizeHint = (bytes: number | undefined): void => { + bufferPoolSizeHint = bytes; +}; + /** * Resolve the `bufferManagerSize` passed to every `new lbug.Database(...)`. - * `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides the default; `0` is a + * `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides everything; `0` is a * deliberate escape hatch that restores LadybugDB's native unbounded - * 80%-of-RAM default. Resolved at call time (not module load) so tests can - * stub the env var and `os.totalmem`. + * 80%-of-RAM default. With no env override, a per-run `setBufferPoolSizeHint` + * (clamped to [floor, default]) sizes the pool to the repo; otherwise the + * default. Resolved at call time (not module load) so tests can stub the env + * var, the hint, and `os.totalmem`. */ const resolveBufferManagerSize = (): number => { const raw = process.env.GITNEXUS_LBUG_BUFFER_POOL_SIZE; - if (raw === undefined) return defaultBufferPoolSize(); + if (raw === undefined) { + return bufferPoolSizeHint !== undefined + ? clampBufferPool(bufferPoolSizeHint) + : defaultBufferPoolSize(); + } const parsed = parseBufferPoolSize(raw); if (parsed !== undefined) return parsed; // Non-empty but unparseable input: warn the operator and fall back — diff --git a/gitnexus/test/unit/lbug-config-wal.test.ts b/gitnexus/test/unit/lbug-config-wal.test.ts index a9912b9ba..4b6d8fe61 100644 --- a/gitnexus/test/unit/lbug-config-wal.test.ts +++ b/gitnexus/test/unit/lbug-config-wal.test.ts @@ -1,9 +1,11 @@ import os from 'os'; -import { describe, expect, it, vi } from 'vitest'; +import { afterEach, describe, expect, it, vi } from 'vitest'; import { createLbugDatabase, + estimateBufferPool, isLbugCheckpointIoError, isWalCorruptionError, + setBufferPoolSizeHint, } from '../../src/core/lbug/lbug-config.js'; import { _captureLogger } from '../../src/core/logger.js'; @@ -252,6 +254,73 @@ describe('createLbugDatabase buffer pool size (#2557)', () => { }); }); +describe('adaptive buffer pool hint', () => { + const GiB = 1024 * 1024 * 1024; + const MiB = 1024 * 1024; + const bufferPoolArg = (Database: ReturnType): unknown => Database.mock.calls[0][1]; + + afterEach(() => setBufferPoolSizeHint(undefined)); + + describe('estimateBufferPool', () => { + it.each([ + ['tiny graph clamps up to the 64 MiB floor', 41, 64 * MiB], + ['mid graph scales linearly (100k elements * 4 KiB = 400 MiB)', 100_000, 100_000 * 4 * 1024], + ['huge graph caps at the 2 GiB / 80%-RAM default', 10_000_000, 2 * GiB], + ])('%s', (_label, elements, expected) => { + const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB); + try { + expect(estimateBufferPool(elements)).toBe(expected); + } finally { + totalmemSpy.mockRestore(); + } + }); + }); + + it.each([ + ['a hint within range passes through', 128 * MiB, 128 * MiB], + ['a hint below the floor clamps up to 64 MiB', 1 * MiB, 64 * MiB], + ['a hint above the default clamps down to the 2 GiB cap', 8 * GiB, 2 * GiB], + ])( + 'createLbugDatabase uses the clamped hint when no env override is set: %s', + (_label, hint, expected) => { + const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB); + try { + setBufferPoolSizeHint(hint); + const Database = vi.fn(function (this: any) {}); + createLbugDatabase({ Database } as any, '/tmp/lbug-hint'); + expect(bufferPoolArg(Database)).toBe(expected); + } finally { + totalmemSpy.mockRestore(); + } + }, + ); + + it('env override wins over the hint (including 0 = native default)', () => { + try { + setBufferPoolSizeHint(128 * MiB); + vi.stubEnv('GITNEXUS_LBUG_BUFFER_POOL_SIZE', '0'); + const Database = vi.fn(function (this: any) {}); + createLbugDatabase({ Database } as any, '/tmp/lbug-hint-env'); + expect(bufferPoolArg(Database)).toBe(0); + } finally { + vi.unstubAllEnvs(); + } + }); + + it('falls back to the default when the hint is cleared', () => { + const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB); + try { + setBufferPoolSizeHint(128 * MiB); + setBufferPoolSizeHint(undefined); + const Database = vi.fn(function (this: any) {}); + createLbugDatabase({ Database } as any, '/tmp/lbug-hint-cleared'); + expect(bufferPoolArg(Database)).toBe(2 * GiB); + } finally { + totalmemSpy.mockRestore(); + } + }); +}); + // ─── Finding 8: strict + permissive checkpoint IO matchers ───────────────── describe('isLbugCheckpointIoError', () => { it.each([ From 087b44a8658f6332bbd113f90de0b94554511802 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 15:51:54 +0000 Subject: [PATCH 02/10] feat(analyze): size the buffer pool to the graph before the DB open runFullAnalysis now sets the buffer-pool size hint from the built graph's node+relationship count (after the pipeline, before initLbug), and clears it at the top of each run so a prior run's size can't leak into a pre-pipeline open. A small repo opens with the fast 64MiB floor instead of eagerly committing the full 2GiB pool. Measured on a 3-file repo (this box): analyze drops 4.78s -> 1.9s for both fresh and incremental, matching a forced 64MiB pool; the 2GiB path still reproduces the old 4.78s. This roughly halves the skills-e2e Idempotency hook (two analyze passes) that was timing out on Windows CI. Co-Authored-By: Claude Fable 5 --- gitnexus/src/core/run-analyze.ts | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 635b255a9..812dc5757 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -34,6 +34,7 @@ import { LbugWipeError, DELETE_FILES_CHUNK_SIZE, } from './lbug/lbug-adapter.js'; +import { estimateBufferPool, setBufferPoolSizeHint } from './lbug/lbug-config.js'; import { escapeCypherString } from './lbug/cypher-escape.js'; import { buildSearchIndexesOrDegrade, @@ -645,6 +646,11 @@ export async function runFullAnalysis( // and are shared across branches (#2106 KTD7). const { storagePath } = getStoragePaths(repoPath); + // Start each analyze with a clean buffer-pool hint: any pre-pipeline DB open + // (e.g. the embeddings-cache open) falls back to the default until the hint is + // set from the built graph below, so a prior run's size can't leak in. + setBufferPoolSizeHint(undefined); + // Clean up stale KuzuDB files from before the LadybugDB migration. const kuzuResult = await cleanupOldKuzuFiles(storagePath); if (kuzuResult.found && kuzuResult.needsReindex) { @@ -1374,6 +1380,15 @@ export async function runFullAnalysis( await wipeLbugDbFiles(lbugPath); } + // Size the buffer pool to the graph just built by the pipeline (a page cache + // over the on-disk index, which scales with node/edge count). A small repo + // opens with the fast 64 MiB floor instead of eagerly committing the full + // 2 GiB default; a large repo still reaches the cap. env override / no-hint + // paths are unchanged. See resolveBufferManagerSize / estimateBufferPool. + setBufferPoolSizeHint( + estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount), + ); + await initLbug(lbugPath); // Manual WAL checkpoint driver (#1741): periodically drain the WAL From 7f03a4ed2db51cb19e6c45e700a79c1aa505b711 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 15:57:38 +0000 Subject: [PATCH 03/10] docs(lbug): correct the POOL_BYTES_PER_ELEMENT tuning note MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The factor is validated by timing a full `analyze --force` of a large repo, not a build-free bench (the pool is a native eager allocation). Benchmark (GitNexus self, 101k graph elements): adaptive 414MiB pool = 35.3s vs forced 2GiB = 50.8s vs forced 64MiB = 26.5s — the adaptive pool is 31% faster than the old 2GiB default even on a large repo (the eager commit dominates), with no under-sizing thrash. 4KiB/element kept. Co-Authored-By: Claude Fable 5 --- gitnexus/src/core/lbug/lbug-config.ts | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index f32a12898..06ee6b9e0 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -346,8 +346,11 @@ const clampBufferPool = (bytes: number): number => * The pool is only a page cache over the on-disk index, and the index scales * with node/edge count — so a per-element budget lets a small repo run with a * small, fast-to-commit pool while a large repo still reaches the 2 GiB cap. - * Kept deliberately generous (holds the working set without thrash); tuned by - * `bench/buffer-pool/measure.mjs` against a large-repo analyze. + * Kept deliberately generous (holds the working set without thrash): tuned by + * timing a full `analyze --force` of a large repo at this factor vs a forced + * 2 GiB pool and confirming no wall-time regression (the pool is a native + * eager allocation, so it is measured with a real analyze, not a build-free + * bench — see the emit-path `COPY` timing note in bench/emit-persistence). */ const POOL_BYTES_PER_ELEMENT = 4 * 1024; From 05dadb79504d20b326c99a33f9f04b69b05e8d9d Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 16:13:26 +0000 Subject: [PATCH 04/10] fix(lbug): raise the adaptive pool floor to a COPY-safe 256 MiB MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 64 MiB floor was too small: LadybugDB's bulk COPY needs working buffer-pool memory that scales with the repo, so a 64 MiB pool fails with "buffer pool is full and no memory could be freed" on any non-trivial repo (empirically: the 6-file skills-e2e idempotency fixture needs >=128 MiB; the ~1800-file GitNexus checkout needs >=256 MiB). Introduce a distinct ADAPTIVE_POOL_FLOOR (256 MiB) for the hint clamp, kept separate from BUFFER_POOL_FLOOR (64 MiB), which still guards defaultBufferPoolSize on tiny-RAM machines; the hint is still clamped up to the machine default so it can never over-commit. This keeps the change as what it actually is — a large-repo optimization: GitNexus full analyze is 51s (2 GiB) -> 35s (adaptive ~414 MiB). Small repos now open COPY-safely at 256 MiB instead of the 2 GiB default (same wall time on Linux, where commit is lazy; the eager commit is far cheaper than 2 GiB on Windows). The reframed comments drop the earlier unrepresentative "3-file repo / 64 MiB fast" claim. Co-Authored-By: Claude Fable 5 --- gitnexus/src/core/lbug/lbug-config.ts | 56 ++++++++++++++-------- gitnexus/src/core/run-analyze.ts | 9 ++-- gitnexus/test/unit/lbug-config-wal.test.ts | 7 +-- 3 files changed, 46 insertions(+), 26 deletions(-) diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index 06ee6b9e0..7b4ba4c65 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -321,6 +321,16 @@ const resolveCheckpointThreshold = (): number => { const DEFAULT_BUFFER_POOL_CAP = 2 * 1024 * 1024 * 1024; const BUFFER_POOL_FLOOR = 64 * 1024 * 1024; +// COPY-safety floor for the adaptive hint (below). LadybugDB's bulk COPY needs +// working buffer-pool memory that scales with the repo: a 64 MiB pool fails +// ("buffer pool is full and no memory could be freed") on any non-trivial repo, +// and even the ~1800-file GitNexus checkout needs ≥256 MiB. So the adaptive +// size never drops a repo below this — a distinct, higher floor than +// BUFFER_POOL_FLOOR, which only guards defaultBufferPoolSize on tiny-RAM +// machines. It is still clamped up to defaultBufferPoolSize, so a machine whose +// default is below this floor keeps its default rather than over-committing. +const ADAPTIVE_POOL_FLOOR = 256 * 1024 * 1024; + const parseBufferPoolSize = (raw: string | undefined): number | undefined => { if (raw === undefined) return undefined; const normalized = raw.trim(); @@ -333,41 +343,49 @@ const parseBufferPoolSize = (raw: string | undefined): number | undefined => { const defaultBufferPoolSize = (): number => Math.min(DEFAULT_BUFFER_POOL_CAP, Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8))); -/** Clamp a requested pool size to [floor, default] — never above the 2 GiB / 80%-RAM default. */ +/** + * Clamp an adaptive pool request to [ADAPTIVE_POOL_FLOOR, default]. The lower + * bound keeps LadybugDB's COPY viable; the upper bound (defaultBufferPoolSize) + * means the hint can only shrink the pool from today's default and can never + * exceed the 2 GiB / 80%-RAM cap — and on a machine whose default is below the + * COPY floor, the default wins, so the pool is never over-committed. + */ const clampBufferPool = (bytes: number): number => - Math.min(defaultBufferPoolSize(), Math.max(BUFFER_POOL_FLOOR, Math.floor(bytes))); + Math.min(defaultBufferPoolSize(), Math.max(ADAPTIVE_POOL_FLOOR, Math.floor(bytes))); /** * Buffer-pool bytes to provision per graph element (node + relationship). * - * LadybugDB eagerly commits the buffer pool at DB open, so an oversized pool - * costs a fixed penalty on every analyze regardless of repo size (measured: - * ~2.8 s extra for a pool ≥128 MiB vs the 64 MiB floor, on a 3-file repo). - * The pool is only a page cache over the on-disk index, and the index scales - * with node/edge count — so a per-element budget lets a small repo run with a - * small, fast-to-commit pool while a large repo still reaches the 2 GiB cap. - * Kept deliberately generous (holds the working set without thrash): tuned by - * timing a full `analyze --force` of a large repo at this factor vs a forced - * 2 GiB pool and confirming no wall-time regression (the pool is a native - * eager allocation, so it is measured with a real analyze, not a build-free - * bench — see the emit-path `COPY` timing note in bench/emit-persistence). + * The fixed 2 GiB default is far larger than most repos' working set, and + * LadybugDB eagerly commits the pool at DB open — measured: a full + * `analyze --force` of the GitNexus checkout takes ~51 s with the 2 GiB pool + * vs ~35 s with the ~414 MiB this factor yields (31% faster; the oversized + * pool's commit dominates). The pool is a page cache over the on-disk index, + * which scales with node/edge count, so a per-element budget sizes it to the + * repo. Kept generous so the whole index stays resident (no COPY thrash) and + * always clamped to at least ADAPTIVE_POOL_FLOOR; tuned by timing a real + * large-repo `analyze --force` at this factor vs a forced 2 GiB pool (the pool + * is a native eager allocation, measured with a real analyze, not a build-free + * bench — see the emit-path COPY timing note in bench/emit-persistence). */ const POOL_BYTES_PER_ELEMENT = 4 * 1024; /** * Size the buffer pool to an estimated graph size (node + relationship count), - * clamped to [BUFFER_POOL_FLOOR, defaultBufferPoolSize()]. The estimate can - * only *shrink* the pool from the default — it never exceeds the 2 GiB / 80%-RAM - * cap — so a large repo is never under the default it gets today. + * clamped to [ADAPTIVE_POOL_FLOOR, defaultBufferPoolSize()]. The estimate can + * only *shrink* the pool from the default — never above the 2 GiB / 80%-RAM cap, + * never below the COPY-safety floor — so no repo is under-sized or gets more + * than the default it would have today. */ export const estimateBufferPool = (graphElementCount: number): number => clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT); /** * Optional per-run buffer-pool size hint (bytes). The analyze orchestrator sets - * it once the graph size is known (after the pipeline, before the DB open) so a - * small repo opens with a small pool, and clears it at run end. Non-analyze - * opens (MCP serve, `native-check` `:memory:`) never set it and keep the default. + * it once the graph size is known (after the pipeline, before the DB open) so + * the pool is sized to the repo instead of the fixed 2 GiB default, and clears + * it at run end. Non-analyze opens (MCP serve, `native-check` `:memory:`) never + * set it and keep the default. */ let bufferPoolSizeHint: number | undefined; diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 812dc5757..ef77ba3d9 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -1381,10 +1381,11 @@ export async function runFullAnalysis( } // Size the buffer pool to the graph just built by the pipeline (a page cache - // over the on-disk index, which scales with node/edge count). A small repo - // opens with the fast 64 MiB floor instead of eagerly committing the full - // 2 GiB default; a large repo still reaches the cap. env override / no-hint - // paths are unchanged. See resolveBufferManagerSize / estimateBufferPool. + // over the on-disk index, which scales with node/edge count) instead of the + // fixed 2 GiB default, whose eager commit dominates large-repo analyze. The + // size is clamped to [COPY-safety floor, default], so it only ever shrinks + // the pool; env override / no-hint paths are unchanged. See + // resolveBufferManagerSize / estimateBufferPool. setBufferPoolSizeHint( estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount), ); diff --git a/gitnexus/test/unit/lbug-config-wal.test.ts b/gitnexus/test/unit/lbug-config-wal.test.ts index 4b6d8fe61..94395541a 100644 --- a/gitnexus/test/unit/lbug-config-wal.test.ts +++ b/gitnexus/test/unit/lbug-config-wal.test.ts @@ -263,7 +263,8 @@ describe('adaptive buffer pool hint', () => { describe('estimateBufferPool', () => { it.each([ - ['tiny graph clamps up to the 64 MiB floor', 41, 64 * MiB], + ['tiny graph clamps up to the 256 MiB COPY-safety floor', 41, 256 * MiB], + ['a graph under the floor still clamps up to 256 MiB', 40_000, 256 * MiB], ['mid graph scales linearly (100k elements * 4 KiB = 400 MiB)', 100_000, 100_000 * 4 * 1024], ['huge graph caps at the 2 GiB / 80%-RAM default', 10_000_000, 2 * GiB], ])('%s', (_label, elements, expected) => { @@ -277,8 +278,8 @@ describe('adaptive buffer pool hint', () => { }); it.each([ - ['a hint within range passes through', 128 * MiB, 128 * MiB], - ['a hint below the floor clamps up to 64 MiB', 1 * MiB, 64 * MiB], + ['a hint within range passes through', 512 * MiB, 512 * MiB], + ['a hint below the COPY-safety floor clamps up to 256 MiB', 100 * MiB, 256 * MiB], ['a hint above the default clamps down to the 2 GiB cap', 8 * GiB, 2 * GiB], ])( 'createLbugDatabase uses the clamped hint when no env override is set: %s', From 8fd1f8a8d839c55337f203f2d6f0ca6ed85aed4b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 20 Jul 2026 19:54:54 +0100 Subject: [PATCH 05/10] test(skills-e2e): give the Idempotency setup hook the 120s budget its siblings use (#2583) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Idempotency beforeAll runs runSkillsCli (analyze --skills) twice, each capped at 45s, under a 90s hook budget — exactly 2x the per-call timeout, with no headroom for fixture creation and git init. On slow Windows CI runners the two analyzes plus setup exceed 90s and the hook times out ('Hook timed out in 90000ms'), failing the shard before the test's own status===null timeout tolerance can apply. Every other describe hook in this file already uses 120s; align this one. Co-authored-by: Claude Co-authored-by: Claude Fable 5 --- gitnexus/test/integration/skills-e2e.test.ts | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/gitnexus/test/integration/skills-e2e.test.ts b/gitnexus/test/integration/skills-e2e.test.ts index 77ecc922d..581497880 100644 --- a/gitnexus/test/integration/skills-e2e.test.ts +++ b/gitnexus/test/integration/skills-e2e.test.ts @@ -2371,7 +2371,13 @@ export function createEntry(level: string, msg: string) { }); result1 = runSkillsCli(tmpDir); result2 = runSkillsCli(tmpDir); - }, 90000); + // 120s to match the other describe hooks in this file. This hook runs + // runSkillsCli TWICE, each capped at 45s, so a 90s budget has no headroom + // over two worst-case analyzes plus fixture setup and git init — it times + // out the *hook* on slow Windows runners (the test below already tolerates + // an individual analyze hitting its own 45s timeout via status === null, + // but a hook timeout fails before that tolerance can apply). + }, 120000); afterAll(() => { fs.rmSync(tmpDir, { recursive: true, force: true }); From 52b06b2642e447666f4b94160a1010bd8c839a4b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 20 Jul 2026 21:26:11 +0100 Subject: [PATCH 06/10] fix(eval): stop using --bare for arms that need Skill or MCP tools (#2584) --bare hard-disables the Skill tool and every mcp__* tool by Claude Code design (confirmed against the pinned 2.1.214 binary; --allowedTools cannot restore what --bare removes). Every workflow_bench arm except baseline_nomcp needs Skill and/or GitNexus MCP tools, so every one of those sessions has been silently unable to invoke gitnexus-plan/work/ review or the CE comparator skills -- the last skill-evolution run (gen 0) scored 0/3 resolution on both arms across every task with error_kind "skill-not-invoked", not because the candidate was bad but because the harness could never invoke either arm's skill at all. Only baseline_nomcp keeps --bare (it explicitly wants zero Skill/MCP access anyway). The rest drop --bare and rely on ANTHROPIC_API_KEY alone; the sandboxed HOME has no OAuth/keychain state to conflict with it, and there's no committed .claude/settings.json in this repo for dropping --bare to newly pick up. Outside --bare the built-in toolset defaults to everything (WebFetch, Task, subagents, ...), and --allowedTools only pre-approves within whatever's available -- it doesn't narrow it. Added --tools for non-bare sessions so the intended tool scope is still enforced instead of silently widening. Claude-Session: https://claude.ai/code/session_01Va5uu9Ar3e45QZ5xFsG4AZ Co-authored-by: Claude Sonnet 5 --- eval/tests/test_workflow_bench_sessions.py | 59 ++++++++++++++++++++++ eval/workflow_bench/runner.py | 9 +++- eval/workflow_bench/runner_sessions.py | 7 +++ 3 files changed, 74 insertions(+), 1 deletion(-) diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index c48ea10a1..e03104ed3 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -167,6 +167,56 @@ def test_run_claude_forwards_the_named_model_to_every_session(monkeypatch, tmp_p assert captured[captured.index("--model") + 1] == "claude-sonnet-4-20250514" +def test_run_claude_restricts_tools_via_tools_flag_outside_bare(monkeypatch, tmp_path): + # Outside --bare, the built-in toolset defaults to everything (subagents, + # WebFetch, Task, ...) and --allowedTools only pre-approves within that — + # it does not narrow it. --tools is what actually restricts the set, so a + # non-bare arm session must pass it or it silently gets a far wider + # toolset than intended. + captured: list[str] = [] + + def fake_run(command, **kwargs): + captured.extend(command) + return fake_cli_result(VALID_REPORT) + + monkeypatch.setattr(runner_sessions, "run_managed", fake_run) + runner.run_claude( + "task", + tmp_path, + claude_bin="claude", + timeout=5, + bare=False, + allowed_tools=["Read", "Edit", "Bash", "Skill"], + ) + tools_idx = captured.index("--tools") + assert captured[tools_idx + 1 : tools_idx + 5] == ["Read", "Edit", "Bash", "Skill"] + allowed_idx = captured.index("--allowedTools") + assert captured[allowed_idx + 1 : allowed_idx + 5] == ["Read", "Edit", "Bash", "Skill"] + + +def test_run_claude_omits_tools_flag_under_bare(monkeypatch, tmp_path): + # --bare already hard-restricts to Bash/Edit/Read on its own (a Claude + # Code design choice, not something --tools/--allowedTools can widen or + # narrow further), so bare sessions must not also pass --tools. + captured: list[str] = [] + + def fake_run(command, **kwargs): + captured.extend(command) + return fake_cli_result(VALID_REPORT) + + monkeypatch.setattr(runner_sessions, "run_managed", fake_run) + runner.run_claude( + "task", + tmp_path, + claude_bin="claude", + timeout=5, + bare=True, + allowed_tools=["Read", "Edit", "Bash", "Skill"], + ) + assert "--tools" not in captured + assert "--allowedTools" in captured + + @pytest.mark.parametrize( ("proc", "expected_kind"), [ @@ -282,6 +332,15 @@ def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, t assert captured[3]["mcp_config_json"] == '{"mcpServers":{}}' assert captured[3]["disallowed_tools"] == ["Skill", "mcp__gitnexus"] + # --bare hard-disables the Skill tool and every mcp__* tool regardless of + # --allowedTools (a Claude Code design choice, not something the harness + # can override) -- every arm here except baseline_nomcp needs Skill + # and/or MCP tools, so only baseline_nomcp may still run under --bare. + assert captured[0]["bare"] is False # workflow: planning session + assert captured[1]["bare"] is False # review + assert captured[2]["bare"] is False # workflow_direct + assert captured[3]["bare"] is True # baseline_nomcp + def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tmp_path): runtime = tmp_path / "gitnexus" diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index cd760eb98..2a180f43f 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -402,6 +402,13 @@ def run_arm( auth_token=args.auth_token, base_url=args.base_url, ) + # --bare hard-disables the Skill tool and every mcp__* tool — by Claude + # Code design, not a bug (--allowedTools can't restore what --bare + # removes). Every arm except baseline_nomcp needs Skill and/or MCP tools, + # so only baseline_nomcp can keep --bare's tighter isolation; the rest + # rely on ANTHROPIC_API_KEY alone (the sandboxed HOME has no OAuth/ + # keychain state to conflict with it). + bare = arm == "baseline_nomcp" common = { "claude_bin": sandbox.claude_bin, "timeout": args.timeout, @@ -412,7 +419,7 @@ def run_arm( read_only_paths=_evaluated_skill_roots(worktree, arm), ), "require_pid_namespace": True, - "bare": True, + "bare": bare, "settings_json": sandbox.settings_json, "strict_mcp_config": True, "mcp_config_json": sandbox_mcp_config(), diff --git a/eval/workflow_bench/runner_sessions.py b/eval/workflow_bench/runner_sessions.py index 60e92b6bc..53e7335ee 100644 --- a/eval/workflow_bench/runner_sessions.py +++ b/eval/workflow_bench/runner_sessions.py @@ -377,6 +377,13 @@ def run_claude( if strict_mcp_config: cmd += ["--strict-mcp-config", "--mcp-config", mcp_config_json or '{"mcpServers":{}}'] if allowed_tools: + # --bare's own hard-coded Bash/Edit/Read ceiling already scopes bare + # sessions; outside --bare the built-in toolset defaults to + # everything (subagents, WebFetch, Task, ...), so --tools is needed + # to actually restrict it — --allowedTools only pre-approves within + # whatever set is available, it does not narrow that set. + if not bare: + cmd += ["--tools", *allowed_tools] cmd += ["--allowedTools", *allowed_tools] if disable_slash_commands: cmd.append("--disable-slash-commands") From 3f93bf22d6765fe5bc61268ea20fd4f0ca91de57 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 21 Jul 2026 05:44:25 +0100 Subject: [PATCH 07/10] chore(deps)(deps): bump js-yaml from 4.3.0 to 5.0.0 in /gitnexus (#2586) Bumps [js-yaml](https://github.com/nodeca/js-yaml) from 4.3.0 to 5.0.0. - [Changelog](https://github.com/nodeca/js-yaml/blob/master/CHANGELOG.md) - [Commits](https://github.com/nodeca/js-yaml/compare/4.3.0...5.0.0) --- updated-dependencies: - dependency-name: js-yaml dependency-version: 5.0.0 dependency-type: direct:production update-type: version-update:semver-major ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 10 +++++----- gitnexus/package.json | 2 +- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 9aa2ef8da..174b31e55 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -24,7 +24,7 @@ "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", "ignore": "^7.0.5", - "js-yaml": "^4.1.1", + "js-yaml": "^5.0.0", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", "node-addon-api": "^8.0.0", @@ -3589,9 +3589,9 @@ "license": "MIT" }, "node_modules/js-yaml": { - "version": "4.3.0", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.0.tgz", - "integrity": "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q==", + "version": "5.0.0", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.0.0.tgz", + "integrity": "sha512-GSvaPUbk1U+FMZ7rJzF+F8e5YVtu7KnD40et/5rBXXRBv2jCO9L3qCewvIDDdudC0QycTFlf6EAA+h3kxBsuUw==", "funding": [ { "type": "github", @@ -3607,7 +3607,7 @@ "argparse": "^2.0.1" }, "bin": { - "js-yaml": "bin/js-yaml.js" + "js-yaml": "bin/js-yaml.mjs" } }, "node_modules/jsesc": { diff --git a/gitnexus/package.json b/gitnexus/package.json index fb60f0dbe..696c74dad 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -70,7 +70,7 @@ "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", "ignore": "^7.0.5", - "js-yaml": "^4.1.1", + "js-yaml": "^5.0.0", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", "node-addon-api": "^8.0.0", From 59359210794efdf36b4589254a949fb20814c752 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 21 Jul 2026 05:44:37 +0100 Subject: [PATCH 08/10] chore(deps)(deps-dev): bump @types/node in /gitnexus (#2588) Bumps [@types/node](https://github.com/DefinitelyTyped/DefinitelyTyped/tree/HEAD/types/node) from 25.9.5 to 26.0.0. - [Release notes](https://github.com/DefinitelyTyped/DefinitelyTyped/releases) - [Commits](https://github.com/DefinitelyTyped/DefinitelyTyped/commits/HEAD/types/node) --- updated-dependencies: - dependency-name: "@types/node" dependency-version: 26.0.0 dependency-type: direct:development update-type: version-update:semver-major ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 16 ++++++++-------- gitnexus/package.json | 2 +- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 174b31e55..882a99731 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -59,7 +59,7 @@ "@types/cors": "^2.8.17", "@types/express": "^5.0.6", "@types/js-yaml": "^4.0.9", - "@types/node": "^25.6.0", + "@types/node": "^26.0.0", "@types/uuid": "^11.0.0", "@vitest/coverage-v8": "^4.0.18", "gitnexus-shared": "file:../gitnexus-shared", @@ -1946,13 +1946,13 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "25.9.5", - "resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.5.tgz", - "integrity": "sha512-OScDchr2fwuUmWdf4kZ9h7PcJiYDVInhJizG/biAq3cAvqwYktuy/TYGGdZNMtNTFUP7rnb0NU4TUdm82kt4Rg==", + "version": "26.0.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.0.0.tgz", + "integrity": "sha512-vf2YFi1iY9lHGwNJMs01biZFbKJkrZR1T6/MlzjhJLPdntOHLhTrDSnSVcdtvjihi4VQNlrFRIxLsDBlQpAipA==", "devOptional": true, "license": "MIT", "dependencies": { - "undici-types": ">=7.24.0 <7.24.7" + "undici-types": "~8.3.0" } }, "node_modules/@types/qs": { @@ -5510,9 +5510,9 @@ } }, "node_modules/undici-types": { - "version": "7.24.6", - "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz", - "integrity": "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==", + "version": "8.3.0", + "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz", + "integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==", "devOptional": true, "license": "MIT" }, diff --git a/gitnexus/package.json b/gitnexus/package.json index 696c74dad..fc7e69e2a 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -106,7 +106,7 @@ "@types/cors": "^2.8.17", "@types/express": "^5.0.6", "@types/js-yaml": "^4.0.9", - "@types/node": "^25.6.0", + "@types/node": "^26.0.0", "@types/uuid": "^11.0.0", "@vitest/coverage-v8": "^4.0.18", "gitnexus-shared": "file:../gitnexus-shared", From 7534f53c27829fb35c3f91573fe4b323032d628e Mon Sep 17 00:00:00 2001 From: ChamHerry <51915924+ChamHerry@users.noreply.github.com> Date: Tue, 21 Jul 2026 12:46:20 +0800 Subject: [PATCH 09/10] feat(embeddings): control request-body dimensions via GITNEXUS_EMBEDDING_REQUEST_DIMS (#2574) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(embeddings): support GITNEXUS_EMBEDDING_REQUEST_DIMS=omit What: Honor GITNEXUS_EMBEDDING_REQUEST_DIMS=omit by suppressing the request-body `dimensions` field sent to HTTP embedding backends. Why: Strict OpenAI-compatible backends return vectors in the model's native size but reject an unfamiliar `dimensions` field, breaking `analyze --embeddings` against them. The var was parsed but never propagated, so `omit` was a no-op. How: Add `requestDimensions` to HttpConfig, return it from readConfig, and forward `config.requestDimensions` (not the validation-only `config.dimensions`) to `httpEmbedBatch`. Local dimension checks still use `config.dimensions`. Details: Coexists with the retry/pacing fields introduced upstream; both feature sets are preserved. Default behavior unchanged when REQUEST_DIMS is unset. Impact: gitnexus/src/core/embeddings/http-client.ts; README; unit tests. * fix(embeddings): name GITNEXUS_EMBEDDING_REQUEST_DIMS in its own config error Address the review findings on #2574. What: - A malformed GITNEXUS_EMBEDDING_REQUEST_DIMS now throws an error naming GITNEXUS_EMBEDDING_REQUEST_DIMS, not the sibling GITNEXUS_EMBEDDING_DIMS. - isHttpEmbeddingDimsError recognizes both leads, so the CLI still classifies the REQUEST_DIMS config mistake as a clean config error, not a stack dump. - Tests: numeric-override decoupling (DIMS=1024 validates the response while REQUEST_DIMS=512 is sent in the body), the omit aliases (none/off/false/0), and the malformed-value error path (which also pins the naming fix). - README documents the full accepted values: omit-aliases and integer override. Why: readConfig reused the DIMS error lead for the REQUEST_DIMS branch, so REQUEST_DIMS=garbage misdirected the operator to edit the wrong variable. The feature's actual decoupling and its non-omit inputs had no test coverage. Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: wangxc Co-authored-by: Gergő Magyar Co-authored-by: Claude Opus 4.8 (1M context) --- gitnexus/README.md | 10 ++ gitnexus/src/core/embeddings/http-client.ts | 53 ++++++--- gitnexus/test/unit/http-embedder.test.ts | 112 ++++++++++++++++++++ 3 files changed, 161 insertions(+), 14 deletions(-) diff --git a/gitnexus/README.md b/gitnexus/README.md index abc193a40..dc8ef17e5 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -284,6 +284,7 @@ Set these env vars to use a remote OpenAI-compatible `/v1/embeddings` endpoint i export GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1 export GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5 export GITNEXUS_EMBEDDING_DIMS=1024 # optional, default 384 +export GITNEXUS_EMBEDDING_REQUEST_DIMS=omit # optional: omit "dimensions", or an integer to override it export GITNEXUS_EMBEDDING_API_KEY=your-key # optional, default: "unused" export GITNEXUS_EMBEDDING_MAX_ATTEMPTS=3 # optional, total attempts (1-20) export GITNEXUS_EMBEDDING_RETRY_CAP_MS=5000 # optional, maximum retry delay @@ -291,6 +292,15 @@ export GITNEXUS_EMBEDDING_MIN_INTERVAL_MS=0 # optional, minimum request spacing gitnexus analyze . --embeddings ``` +`GITNEXUS_EMBEDDING_REQUEST_DIMS` controls only the `dimensions` field sent in +the request body, independently of `GITNEXUS_EMBEDDING_DIMS` (which still +validates the returned vector's length): + +- `omit` (or `none`, `off`, `false`, `0`) — do not send `dimensions` at all, for + strict backends that return the right vector size but reject the field. +- a positive integer — send that value instead of `GITNEXUS_EMBEDDING_DIMS`. +- unset — send `GITNEXUS_EMBEDDING_DIMS` (the previous behavior). + Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. Retry and pacing settings are provider-neutral; provider-specific limits should be supplied through configuration. When unset, local embeddings are used unchanged. ## Multi-Repo Support diff --git a/gitnexus/src/core/embeddings/http-client.ts b/gitnexus/src/core/embeddings/http-client.ts index c85d9ff02..f3eb3bb5a 100644 --- a/gitnexus/src/core/embeddings/http-client.ts +++ b/gitnexus/src/core/embeddings/http-client.ts @@ -29,6 +29,7 @@ interface HttpConfig { maxAttempts: number; retryCapMs: number; minIntervalMs: number; + requestDimensions?: number; } export interface EmbeddingRequestOptions { @@ -106,20 +107,26 @@ const paceHttpRequest = async (minIntervalMs: number, signal?: AbortSignal): Pro }; /** - * Stable lead of the {@link readConfig} malformed-`GITNEXUS_EMBEDDING_DIMS` - * error. `readConfig` throws a plain `Error` (not an {@link HttpEmbeddingError}) - * because this is a *config* mistake, not an endpoint failure — so the CLI - * recognizes it by this lead ({@link isHttpEmbeddingDimsError}) and prints a - * clean config message instead of a raw stack dump. See #2385. + * Stable lead of a {@link readConfig} malformed dims-env error. `readConfig` + * throws a plain `Error` (not an {@link HttpEmbeddingError}) for a malformed + * `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS` because it's a + * *config* mistake, not an endpoint failure — so the CLI recognizes it by this + * lead ({@link isHttpEmbeddingDimsError}) and prints a clean config message + * instead of a raw stack dump. Each var names itself so the message points the + * operator at the variable they actually set, not a sibling. See #2385. */ -const EMBEDDING_DIMS_ENV_ERROR_LEAD = 'GITNEXUS_EMBEDDING_DIMS must be a positive integer'; +const dimsEnvErrorLead = (name: string): string => `${name} must be a positive integer`; +const EMBEDDING_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_DIMS'); +const EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_REQUEST_DIMS'); /** - * @internal Exported for the CLI analyze error handler. True when `message` is - * the {@link readConfig} malformed-DIMS config error (a plain `Error`). + * @internal Exported for the CLI analyze error handler. True when `message` is a + * {@link readConfig} malformed dims-env config error (a plain `Error`) — for + * either `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS`. */ export const isHttpEmbeddingDimsError = (message: string): boolean => - message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD); + message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD) || + message.includes(EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD); /** * Build config from the current process.env snapshot. @@ -147,6 +154,23 @@ const readConfig = (): HttpConfig | null => { dimensions = parsed; } + const rawRequestDims = process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS?.trim(); + let requestDimensions = dimensions; + if (rawRequestDims) { + if (/^(omit|none|off|false|0)$/i.test(rawRequestDims)) { + requestDimensions = undefined; + } else { + if (!/^\d+$/.test(rawRequestDims)) { + throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`); + } + const parsed = parseInt(rawRequestDims, 10); + if (parsed <= 0) { + throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`); + } + requestDimensions = parsed; + } + } + return { baseUrl: baseUrl.replace(/\/+$/, ''), model, @@ -163,6 +187,7 @@ const readConfig = (): HttpConfig | null => { 300_000, ), minIntervalMs: parseNonNegativeIntegerEnv('GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', 0, 300_000), + requestDimensions, }; }; @@ -283,9 +308,9 @@ const isEmbeddingItem = (item: unknown): item is EmbeddingItem => * the `dimensions` field in the request body. Endpoints that implement * Matryoshka truncation (OpenAI text-embedding-3-*, Cohere embed-v3, * Voyage) return a truncated vector at that size; endpoints that do not - * recognise the field may ignore it or return 400. Leave - * `GITNEXUS_EMBEDDING_DIMS` unset for strict backends that reject - * unknown fields. + * recognise the field may ignore it or return 400. Set + * `GITNEXUS_EMBEDDING_REQUEST_DIMS=omit` for strict backends while keeping + * `GITNEXUS_EMBEDDING_DIMS` set to the returned vector size. */ const httpEmbedBatch = async ( url: string, @@ -434,7 +459,7 @@ export const httpEmbed = async ( config.model, config.apiKey, batchIndex, - config.dimensions, + config.requestDimensions, requestOptions, config.maxAttempts, config.retryCapMs, @@ -491,7 +516,7 @@ export const httpEmbedQuery = async ( config.model, config.apiKey, 0, - config.dimensions, + config.requestDimensions, requestOptions, config.maxAttempts, config.retryCapMs, diff --git a/gitnexus/test/unit/http-embedder.test.ts b/gitnexus/test/unit/http-embedder.test.ts index 63cd715f9..fb4dde0af 100644 --- a/gitnexus/test/unit/http-embedder.test.ts +++ b/gitnexus/test/unit/http-embedder.test.ts @@ -9,6 +9,7 @@ const ENV_KEYS = [ 'GITNEXUS_EMBEDDING_MAX_ATTEMPTS', 'GITNEXUS_EMBEDDING_RETRY_CAP_MS', 'GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', + 'GITNEXUS_EMBEDDING_REQUEST_DIMS', ] as const; /** 384d mock vector matching the default schema dimensions. */ @@ -166,6 +167,30 @@ describe('HTTP embedding backend', () => { expect(result.length).toBe(1024); }); + it('can validate custom dims without forwarding dimensions to strict backends', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'omit'; + + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect('dimensions' in body).toBe(false); + expect(body.model).toBe('bge-m3'); + expect(result.length).toBe(1024); + }); + it('forwards dimensions on the single-query path', async () => { process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; process.env.GITNEXUS_EMBEDDING_MODEL = 'text-embedding-3-large'; @@ -188,6 +213,93 @@ describe('HTTP embedding backend', () => { expect(result.length).toBe(512); }); + it('can omit dimensions on the single-query path while validating custom dims', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'omit'; + + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const mod = await import('../../src/mcp/core/embedder.js'); + const result = await mod.embedQuery('query text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect('dimensions' in body).toBe(false); + expect(result.length).toBe(1024); + }); + + it.each(['none', 'off', 'false', '0'])( + 'treats GITNEXUS_EMBEDDING_REQUEST_DIMS=%s as omit and drops the request dimensions field', + async (alias) => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = alias; + + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect('dimensions' in body).toBe(false); + expect(result.length).toBe(1024); + }, + ); + + it('sends REQUEST_DIMS as the request dimensions while DIMS validates the response', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'text-embedding-3-large'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = '512'; + + // Response keeps the DIMS-validated length; only the outgoing request differs. + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect(body.dimensions).toBe(512); + expect(result.length).toBe(1024); + }); + + it('rejects a malformed GITNEXUS_EMBEDDING_REQUEST_DIMS with an error naming that var', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'garbage'; + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const { isHttpEmbeddingDimsError } = await import('../../src/core/embeddings/http-client.js'); + const err = await embedText('test').catch((e: unknown) => e); + // Recognizable as a config error so the CLI prints a clean message... + expect(isHttpEmbeddingDimsError(String(err))).toBe(true); + // ...and it points the operator at the var they set, not GITNEXUS_EMBEDDING_DIMS. + expect(String(err)).toContain('GITNEXUS_EMBEDDING_REQUEST_DIMS must be a positive integer'); + }); + it('retries on server error', async () => { process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; From 84b4402cd18a4f551a6eb05d0020fb24a8bdef16 Mon Sep 17 00:00:00 2001 From: ChunxueLi <54129170+ChunxueLi@users.noreply.github.com> Date: Tue, 21 Jul 2026 12:48:55 +0800 Subject: [PATCH 10/10] fix: remove hardcoded 300-flows cap for large repositories (#2198) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix: remove hardcoded 300-flows cap for large repositories The dynamicMaxProcesses was capped at 300 via Math.min(300, ...), causing large repositories (280K+ nodes) to lose execution flows. Change: Remove the Math.min(300, ...) cap, keep dynamic calculation. Effect: 280K-node repo: 300 → 1617 flows. * test: add regression for dynamic maxProcesses sizing (#2198) Verify that processProcesses honours maxProcesses > 300 without truncation. Addresses the optional follow-up suggested by @koriyoshi2041. * test: exercise computeDynamicMaxProcesses at the phase layer (#2198) Extract from the inline expression in so the regression test can exercise the function that actually contained the removed cap. The previous test called directly with , which passes regardless of whether the phase-level cap is present — never had the cap. The new test suite covers: - floor (20) for tiny repos - linear scaling in the 0–3000 range - growth past 300 for large repos (the actual regression) - explicit assertion that reintroducing Math.min(300, …) would fail Addresses review feedback from @azizur100389. --------- Co-authored-by: Ubuntu Co-authored-by: Gergő Magyar --- .../ingestion/pipeline-phases/processes.ts | 13 ++++++- gitnexus/test/unit/process-processor.test.ts | 37 +++++++++++++++++++ 2 files changed, 49 insertions(+), 1 deletion(-) diff --git a/gitnexus/src/core/ingestion/pipeline-phases/processes.ts b/gitnexus/src/core/ingestion/pipeline-phases/processes.ts index fd2f25cc8..462d1523f 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/processes.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/processes.ts @@ -25,6 +25,17 @@ export interface ProcessesOutput { processResult: ProcessDetectionResult; } +/** + * Compute the dynamic max-processes budget from the symbol count. + * + * Scales proportionally (symbolCount / 10) with a floor of 20. + * Prior to #2198 this was capped at 300 via `Math.min(300, …)`, + * silently truncating process detection on large repositories. + */ +export function computeDynamicMaxProcesses(symbolCount: number): number { + return Math.max(20, Math.round(symbolCount / 10)); +} + export const processesPhase: PipelinePhase = { name: 'processes', // `structure` supplies `totalFiles` (progress counter) without the spurious @@ -53,7 +64,7 @@ export const processesPhase: PipelinePhase = { ctx.graph.forEachNode((n) => { if (n.label !== 'File') symbolCount++; }); - const dynamicMaxProcesses = Math.max(20, Math.min(300, Math.round(symbolCount / 10))); + const dynamicMaxProcesses = computeDynamicMaxProcesses(symbolCount); const processResult = await processProcesses( ctx.graph, diff --git a/gitnexus/test/unit/process-processor.test.ts b/gitnexus/test/unit/process-processor.test.ts index 18d5c7be3..09600c6de 100644 --- a/gitnexus/test/unit/process-processor.test.ts +++ b/gitnexus/test/unit/process-processor.test.ts @@ -3,6 +3,7 @@ import { processProcesses, type ProcessDetectionConfig, } from '../../src/core/ingestion/process-processor.js'; +import { computeDynamicMaxProcesses } from '../../src/core/ingestion/pipeline-phases/processes.js'; import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; import type { CommunityMembership } from '../../src/core/ingestion/community-processor.js'; @@ -522,4 +523,40 @@ describe('processProcesses', () => { expect(result.processes.length).toBeLessThanOrEqual(3); expect(result.stats.totalProcesses).toBeLessThanOrEqual(3); }); + + // Regression for #2198: the processesPhase dynamic sizing used to cap at + // Math.min(300, symbolCount/10). On large repos (>3000 symbols) that silently + // truncated the process index. The cap was removed by extracting + // computeDynamicMaxProcesses() — this test exercises the helper directly + // so it fails if someone reintroduces the 300 ceiling. + describe('computeDynamicMaxProcesses (#2198)', () => { + it('returns at least the floor of 20 for tiny repos', () => { + expect(computeDynamicMaxProcesses(0)).toBe(20); + expect(computeDynamicMaxProcesses(50)).toBe(20); // 50/10 = 5, floored to 20 + expect(computeDynamicMaxProcesses(199)).toBe(20); // 199/10 ≈ 20 + }); + + it('scales linearly within the old 0–3000 range', () => { + expect(computeDynamicMaxProcesses(500)).toBe(50); + expect(computeDynamicMaxProcesses(1000)).toBe(100); + expect(computeDynamicMaxProcesses(2999)).toBe(300); + }); + + it('grows past 300 for large repos — the regression that #2198 fixes', () => { + // 3001 symbols → 300 (just at the boundary) + expect(computeDynamicMaxProcesses(3001)).toBe(300); + // 3100 symbols → 310 — would have been capped to 300 before the fix + expect(computeDynamicMaxProcesses(3100)).toBe(310); + // 5000 symbols → 500 + expect(computeDynamicMaxProcesses(5000)).toBe(500); + // 28000 symbols (real-world large repo) → 2800 + expect(computeDynamicMaxProcesses(28000)).toBe(2800); + }); + + it('does NOT cap at 300 — fails if Math.min(300, ...) is reintroduced', () => { + const largeRepo = computeDynamicMaxProcesses(10000); + expect(largeRepo).toBe(1000); + expect(largeRepo).toBeGreaterThan(300); + }); + }); });