diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index c48ea10a1..e03104ed3 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -167,6 +167,56 @@ def test_run_claude_forwards_the_named_model_to_every_session(monkeypatch, tmp_p assert captured[captured.index("--model") + 1] == "claude-sonnet-4-20250514" +def test_run_claude_restricts_tools_via_tools_flag_outside_bare(monkeypatch, tmp_path): + # Outside --bare, the built-in toolset defaults to everything (subagents, + # WebFetch, Task, ...) and --allowedTools only pre-approves within that — + # it does not narrow it. --tools is what actually restricts the set, so a + # non-bare arm session must pass it or it silently gets a far wider + # toolset than intended. + captured: list[str] = [] + + def fake_run(command, **kwargs): + captured.extend(command) + return fake_cli_result(VALID_REPORT) + + monkeypatch.setattr(runner_sessions, "run_managed", fake_run) + runner.run_claude( + "task", + tmp_path, + claude_bin="claude", + timeout=5, + bare=False, + allowed_tools=["Read", "Edit", "Bash", "Skill"], + ) + tools_idx = captured.index("--tools") + assert captured[tools_idx + 1 : tools_idx + 5] == ["Read", "Edit", "Bash", "Skill"] + allowed_idx = captured.index("--allowedTools") + assert captured[allowed_idx + 1 : allowed_idx + 5] == ["Read", "Edit", "Bash", "Skill"] + + +def test_run_claude_omits_tools_flag_under_bare(monkeypatch, tmp_path): + # --bare already hard-restricts to Bash/Edit/Read on its own (a Claude + # Code design choice, not something --tools/--allowedTools can widen or + # narrow further), so bare sessions must not also pass --tools. + captured: list[str] = [] + + def fake_run(command, **kwargs): + captured.extend(command) + return fake_cli_result(VALID_REPORT) + + monkeypatch.setattr(runner_sessions, "run_managed", fake_run) + runner.run_claude( + "task", + tmp_path, + claude_bin="claude", + timeout=5, + bare=True, + allowed_tools=["Read", "Edit", "Bash", "Skill"], + ) + assert "--tools" not in captured + assert "--allowedTools" in captured + + @pytest.mark.parametrize( ("proc", "expected_kind"), [ @@ -282,6 +332,15 @@ def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, t assert captured[3]["mcp_config_json"] == '{"mcpServers":{}}' assert captured[3]["disallowed_tools"] == ["Skill", "mcp__gitnexus"] + # --bare hard-disables the Skill tool and every mcp__* tool regardless of + # --allowedTools (a Claude Code design choice, not something the harness + # can override) -- every arm here except baseline_nomcp needs Skill + # and/or MCP tools, so only baseline_nomcp may still run under --bare. + assert captured[0]["bare"] is False # workflow: planning session + assert captured[1]["bare"] is False # review + assert captured[2]["bare"] is False # workflow_direct + assert captured[3]["bare"] is True # baseline_nomcp + def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tmp_path): runtime = tmp_path / "gitnexus" diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index cd760eb98..2a180f43f 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -402,6 +402,13 @@ def run_arm( auth_token=args.auth_token, base_url=args.base_url, ) + # --bare hard-disables the Skill tool and every mcp__* tool — by Claude + # Code design, not a bug (--allowedTools can't restore what --bare + # removes). Every arm except baseline_nomcp needs Skill and/or MCP tools, + # so only baseline_nomcp can keep --bare's tighter isolation; the rest + # rely on ANTHROPIC_API_KEY alone (the sandboxed HOME has no OAuth/ + # keychain state to conflict with it). + bare = arm == "baseline_nomcp" common = { "claude_bin": sandbox.claude_bin, "timeout": args.timeout, @@ -412,7 +419,7 @@ def run_arm( read_only_paths=_evaluated_skill_roots(worktree, arm), ), "require_pid_namespace": True, - "bare": True, + "bare": bare, "settings_json": sandbox.settings_json, "strict_mcp_config": True, "mcp_config_json": sandbox_mcp_config(), diff --git a/eval/workflow_bench/runner_sessions.py b/eval/workflow_bench/runner_sessions.py index 60e92b6bc..53e7335ee 100644 --- a/eval/workflow_bench/runner_sessions.py +++ b/eval/workflow_bench/runner_sessions.py @@ -377,6 +377,13 @@ def run_claude( if strict_mcp_config: cmd += ["--strict-mcp-config", "--mcp-config", mcp_config_json or '{"mcpServers":{}}'] if allowed_tools: + # --bare's own hard-coded Bash/Edit/Read ceiling already scopes bare + # sessions; outside --bare the built-in toolset defaults to + # everything (subagents, WebFetch, Task, ...), so --tools is needed + # to actually restrict it — --allowedTools only pre-approves within + # whatever set is available, it does not narrow that set. + if not bare: + cmd += ["--tools", *allowed_tools] cmd += ["--allowedTools", *allowed_tools] if disable_slash_commands: cmd.append("--disable-slash-commands") diff --git a/gitnexus/README.md b/gitnexus/README.md index abc193a40..dc8ef17e5 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -284,6 +284,7 @@ Set these env vars to use a remote OpenAI-compatible `/v1/embeddings` endpoint i export GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1 export GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5 export GITNEXUS_EMBEDDING_DIMS=1024 # optional, default 384 +export GITNEXUS_EMBEDDING_REQUEST_DIMS=omit # optional: omit "dimensions", or an integer to override it export GITNEXUS_EMBEDDING_API_KEY=your-key # optional, default: "unused" export GITNEXUS_EMBEDDING_MAX_ATTEMPTS=3 # optional, total attempts (1-20) export GITNEXUS_EMBEDDING_RETRY_CAP_MS=5000 # optional, maximum retry delay @@ -291,6 +292,15 @@ export GITNEXUS_EMBEDDING_MIN_INTERVAL_MS=0 # optional, minimum request spacing gitnexus analyze . --embeddings ``` +`GITNEXUS_EMBEDDING_REQUEST_DIMS` controls only the `dimensions` field sent in +the request body, independently of `GITNEXUS_EMBEDDING_DIMS` (which still +validates the returned vector's length): + +- `omit` (or `none`, `off`, `false`, `0`) — do not send `dimensions` at all, for + strict backends that return the right vector size but reject the field. +- a positive integer — send that value instead of `GITNEXUS_EMBEDDING_DIMS`. +- unset — send `GITNEXUS_EMBEDDING_DIMS` (the previous behavior). + Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. Retry and pacing settings are provider-neutral; provider-specific limits should be supplied through configuration. When unset, local embeddings are used unchanged. ## Multi-Repo Support diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 9aa2ef8da..882a99731 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -24,7 +24,7 @@ "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", "ignore": "^7.0.5", - "js-yaml": "^4.1.1", + "js-yaml": "^5.0.0", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", "node-addon-api": "^8.0.0", @@ -59,7 +59,7 @@ "@types/cors": "^2.8.17", "@types/express": "^5.0.6", "@types/js-yaml": "^4.0.9", - "@types/node": "^25.6.0", + "@types/node": "^26.0.0", "@types/uuid": "^11.0.0", "@vitest/coverage-v8": "^4.0.18", "gitnexus-shared": "file:../gitnexus-shared", @@ -1946,13 +1946,13 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "25.9.5", - "resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.5.tgz", - "integrity": "sha512-OScDchr2fwuUmWdf4kZ9h7PcJiYDVInhJizG/biAq3cAvqwYktuy/TYGGdZNMtNTFUP7rnb0NU4TUdm82kt4Rg==", + "version": "26.0.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.0.0.tgz", + "integrity": "sha512-vf2YFi1iY9lHGwNJMs01biZFbKJkrZR1T6/MlzjhJLPdntOHLhTrDSnSVcdtvjihi4VQNlrFRIxLsDBlQpAipA==", "devOptional": true, "license": "MIT", "dependencies": { - "undici-types": ">=7.24.0 <7.24.7" + "undici-types": "~8.3.0" } }, "node_modules/@types/qs": { @@ -3589,9 +3589,9 @@ "license": "MIT" }, "node_modules/js-yaml": { - "version": "4.3.0", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.0.tgz", - "integrity": "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q==", + "version": "5.0.0", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.0.0.tgz", + "integrity": "sha512-GSvaPUbk1U+FMZ7rJzF+F8e5YVtu7KnD40et/5rBXXRBv2jCO9L3qCewvIDDdudC0QycTFlf6EAA+h3kxBsuUw==", "funding": [ { "type": "github", @@ -3607,7 +3607,7 @@ "argparse": "^2.0.1" }, "bin": { - "js-yaml": "bin/js-yaml.js" + "js-yaml": "bin/js-yaml.mjs" } }, "node_modules/jsesc": { @@ -5510,9 +5510,9 @@ } }, "node_modules/undici-types": { - "version": "7.24.6", - "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz", - "integrity": "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==", + "version": "8.3.0", + "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz", + "integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==", "devOptional": true, "license": "MIT" }, diff --git a/gitnexus/package.json b/gitnexus/package.json index fb60f0dbe..fc7e69e2a 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -70,7 +70,7 @@ "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", "ignore": "^7.0.5", - "js-yaml": "^4.1.1", + "js-yaml": "^5.0.0", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", "node-addon-api": "^8.0.0", @@ -106,7 +106,7 @@ "@types/cors": "^2.8.17", "@types/express": "^5.0.6", "@types/js-yaml": "^4.0.9", - "@types/node": "^25.6.0", + "@types/node": "^26.0.0", "@types/uuid": "^11.0.0", "@vitest/coverage-v8": "^4.0.18", "gitnexus-shared": "file:../gitnexus-shared", diff --git a/gitnexus/src/core/embeddings/http-client.ts b/gitnexus/src/core/embeddings/http-client.ts index c85d9ff02..f3eb3bb5a 100644 --- a/gitnexus/src/core/embeddings/http-client.ts +++ b/gitnexus/src/core/embeddings/http-client.ts @@ -29,6 +29,7 @@ interface HttpConfig { maxAttempts: number; retryCapMs: number; minIntervalMs: number; + requestDimensions?: number; } export interface EmbeddingRequestOptions { @@ -106,20 +107,26 @@ const paceHttpRequest = async (minIntervalMs: number, signal?: AbortSignal): Pro }; /** - * Stable lead of the {@link readConfig} malformed-`GITNEXUS_EMBEDDING_DIMS` - * error. `readConfig` throws a plain `Error` (not an {@link HttpEmbeddingError}) - * because this is a *config* mistake, not an endpoint failure — so the CLI - * recognizes it by this lead ({@link isHttpEmbeddingDimsError}) and prints a - * clean config message instead of a raw stack dump. See #2385. + * Stable lead of a {@link readConfig} malformed dims-env error. `readConfig` + * throws a plain `Error` (not an {@link HttpEmbeddingError}) for a malformed + * `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS` because it's a + * *config* mistake, not an endpoint failure — so the CLI recognizes it by this + * lead ({@link isHttpEmbeddingDimsError}) and prints a clean config message + * instead of a raw stack dump. Each var names itself so the message points the + * operator at the variable they actually set, not a sibling. See #2385. */ -const EMBEDDING_DIMS_ENV_ERROR_LEAD = 'GITNEXUS_EMBEDDING_DIMS must be a positive integer'; +const dimsEnvErrorLead = (name: string): string => `${name} must be a positive integer`; +const EMBEDDING_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_DIMS'); +const EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_REQUEST_DIMS'); /** - * @internal Exported for the CLI analyze error handler. True when `message` is - * the {@link readConfig} malformed-DIMS config error (a plain `Error`). + * @internal Exported for the CLI analyze error handler. True when `message` is a + * {@link readConfig} malformed dims-env config error (a plain `Error`) — for + * either `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS`. */ export const isHttpEmbeddingDimsError = (message: string): boolean => - message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD); + message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD) || + message.includes(EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD); /** * Build config from the current process.env snapshot. @@ -147,6 +154,23 @@ const readConfig = (): HttpConfig | null => { dimensions = parsed; } + const rawRequestDims = process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS?.trim(); + let requestDimensions = dimensions; + if (rawRequestDims) { + if (/^(omit|none|off|false|0)$/i.test(rawRequestDims)) { + requestDimensions = undefined; + } else { + if (!/^\d+$/.test(rawRequestDims)) { + throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`); + } + const parsed = parseInt(rawRequestDims, 10); + if (parsed <= 0) { + throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`); + } + requestDimensions = parsed; + } + } + return { baseUrl: baseUrl.replace(/\/+$/, ''), model, @@ -163,6 +187,7 @@ const readConfig = (): HttpConfig | null => { 300_000, ), minIntervalMs: parseNonNegativeIntegerEnv('GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', 0, 300_000), + requestDimensions, }; }; @@ -283,9 +308,9 @@ const isEmbeddingItem = (item: unknown): item is EmbeddingItem => * the `dimensions` field in the request body. Endpoints that implement * Matryoshka truncation (OpenAI text-embedding-3-*, Cohere embed-v3, * Voyage) return a truncated vector at that size; endpoints that do not - * recognise the field may ignore it or return 400. Leave - * `GITNEXUS_EMBEDDING_DIMS` unset for strict backends that reject - * unknown fields. + * recognise the field may ignore it or return 400. Set + * `GITNEXUS_EMBEDDING_REQUEST_DIMS=omit` for strict backends while keeping + * `GITNEXUS_EMBEDDING_DIMS` set to the returned vector size. */ const httpEmbedBatch = async ( url: string, @@ -434,7 +459,7 @@ export const httpEmbed = async ( config.model, config.apiKey, batchIndex, - config.dimensions, + config.requestDimensions, requestOptions, config.maxAttempts, config.retryCapMs, @@ -491,7 +516,7 @@ export const httpEmbedQuery = async ( config.model, config.apiKey, 0, - config.dimensions, + config.requestDimensions, requestOptions, config.maxAttempts, config.retryCapMs, diff --git a/gitnexus/src/core/ingestion/pipeline-phases/processes.ts b/gitnexus/src/core/ingestion/pipeline-phases/processes.ts index fd2f25cc8..462d1523f 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/processes.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/processes.ts @@ -25,6 +25,17 @@ export interface ProcessesOutput { processResult: ProcessDetectionResult; } +/** + * Compute the dynamic max-processes budget from the symbol count. + * + * Scales proportionally (symbolCount / 10) with a floor of 20. + * Prior to #2198 this was capped at 300 via `Math.min(300, …)`, + * silently truncating process detection on large repositories. + */ +export function computeDynamicMaxProcesses(symbolCount: number): number { + return Math.max(20, Math.round(symbolCount / 10)); +} + export const processesPhase: PipelinePhase = { name: 'processes', // `structure` supplies `totalFiles` (progress counter) without the spurious @@ -53,7 +64,7 @@ export const processesPhase: PipelinePhase = { ctx.graph.forEachNode((n) => { if (n.label !== 'File') symbolCount++; }); - const dynamicMaxProcesses = Math.max(20, Math.min(300, Math.round(symbolCount / 10))); + const dynamicMaxProcesses = computeDynamicMaxProcesses(symbolCount); const processResult = await processProcesses( ctx.graph, diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index 540fe530a..7b4ba4c65 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -321,6 +321,16 @@ const resolveCheckpointThreshold = (): number => { const DEFAULT_BUFFER_POOL_CAP = 2 * 1024 * 1024 * 1024; const BUFFER_POOL_FLOOR = 64 * 1024 * 1024; +// COPY-safety floor for the adaptive hint (below). LadybugDB's bulk COPY needs +// working buffer-pool memory that scales with the repo: a 64 MiB pool fails +// ("buffer pool is full and no memory could be freed") on any non-trivial repo, +// and even the ~1800-file GitNexus checkout needs ≥256 MiB. So the adaptive +// size never drops a repo below this — a distinct, higher floor than +// BUFFER_POOL_FLOOR, which only guards defaultBufferPoolSize on tiny-RAM +// machines. It is still clamped up to defaultBufferPoolSize, so a machine whose +// default is below this floor keeps its default rather than over-committing. +const ADAPTIVE_POOL_FLOOR = 256 * 1024 * 1024; + const parseBufferPoolSize = (raw: string | undefined): number | undefined => { if (raw === undefined) return undefined; const normalized = raw.trim(); @@ -333,16 +343,73 @@ const parseBufferPoolSize = (raw: string | undefined): number | undefined => { const defaultBufferPoolSize = (): number => Math.min(DEFAULT_BUFFER_POOL_CAP, Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8))); +/** + * Clamp an adaptive pool request to [ADAPTIVE_POOL_FLOOR, default]. The lower + * bound keeps LadybugDB's COPY viable; the upper bound (defaultBufferPoolSize) + * means the hint can only shrink the pool from today's default and can never + * exceed the 2 GiB / 80%-RAM cap — and on a machine whose default is below the + * COPY floor, the default wins, so the pool is never over-committed. + */ +const clampBufferPool = (bytes: number): number => + Math.min(defaultBufferPoolSize(), Math.max(ADAPTIVE_POOL_FLOOR, Math.floor(bytes))); + +/** + * Buffer-pool bytes to provision per graph element (node + relationship). + * + * The fixed 2 GiB default is far larger than most repos' working set, and + * LadybugDB eagerly commits the pool at DB open — measured: a full + * `analyze --force` of the GitNexus checkout takes ~51 s with the 2 GiB pool + * vs ~35 s with the ~414 MiB this factor yields (31% faster; the oversized + * pool's commit dominates). The pool is a page cache over the on-disk index, + * which scales with node/edge count, so a per-element budget sizes it to the + * repo. Kept generous so the whole index stays resident (no COPY thrash) and + * always clamped to at least ADAPTIVE_POOL_FLOOR; tuned by timing a real + * large-repo `analyze --force` at this factor vs a forced 2 GiB pool (the pool + * is a native eager allocation, measured with a real analyze, not a build-free + * bench — see the emit-path COPY timing note in bench/emit-persistence). + */ +const POOL_BYTES_PER_ELEMENT = 4 * 1024; + +/** + * Size the buffer pool to an estimated graph size (node + relationship count), + * clamped to [ADAPTIVE_POOL_FLOOR, defaultBufferPoolSize()]. The estimate can + * only *shrink* the pool from the default — never above the 2 GiB / 80%-RAM cap, + * never below the COPY-safety floor — so no repo is under-sized or gets more + * than the default it would have today. + */ +export const estimateBufferPool = (graphElementCount: number): number => + clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT); + +/** + * Optional per-run buffer-pool size hint (bytes). The analyze orchestrator sets + * it once the graph size is known (after the pipeline, before the DB open) so + * the pool is sized to the repo instead of the fixed 2 GiB default, and clears + * it at run end. Non-analyze opens (MCP serve, `native-check` `:memory:`) never + * set it and keep the default. + */ +let bufferPoolSizeHint: number | undefined; + +/** Set (bytes) or clear (`undefined`) the per-run buffer-pool size hint. */ +export const setBufferPoolSizeHint = (bytes: number | undefined): void => { + bufferPoolSizeHint = bytes; +}; + /** * Resolve the `bufferManagerSize` passed to every `new lbug.Database(...)`. - * `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides the default; `0` is a + * `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides everything; `0` is a * deliberate escape hatch that restores LadybugDB's native unbounded - * 80%-of-RAM default. Resolved at call time (not module load) so tests can - * stub the env var and `os.totalmem`. + * 80%-of-RAM default. With no env override, a per-run `setBufferPoolSizeHint` + * (clamped to [floor, default]) sizes the pool to the repo; otherwise the + * default. Resolved at call time (not module load) so tests can stub the env + * var, the hint, and `os.totalmem`. */ const resolveBufferManagerSize = (): number => { const raw = process.env.GITNEXUS_LBUG_BUFFER_POOL_SIZE; - if (raw === undefined) return defaultBufferPoolSize(); + if (raw === undefined) { + return bufferPoolSizeHint !== undefined + ? clampBufferPool(bufferPoolSizeHint) + : defaultBufferPoolSize(); + } const parsed = parseBufferPoolSize(raw); if (parsed !== undefined) return parsed; // Non-empty but unparseable input: warn the operator and fall back — diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index e0bbfa116..c71a716d8 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -34,6 +34,7 @@ import { LbugWipeError, DELETE_FILES_CHUNK_SIZE, } from './lbug/lbug-adapter.js'; +import { estimateBufferPool, setBufferPoolSizeHint } from './lbug/lbug-config.js'; import { escapeCypherString } from './lbug/cypher-escape.js'; import { buildSearchIndexesOrDegrade, @@ -647,6 +648,11 @@ export async function runFullAnalysis( // and are shared across branches (#2106 KTD7). const { storagePath } = getStoragePaths(repoPath); + // Start each analyze with a clean buffer-pool hint: any pre-pipeline DB open + // (e.g. the embeddings-cache open) falls back to the default until the hint is + // set from the built graph below, so a prior run's size can't leak in. + setBufferPoolSizeHint(undefined); + // Clean up stale KuzuDB files from before the LadybugDB migration. const kuzuResult = await cleanupOldKuzuFiles(storagePath); if (kuzuResult.found && kuzuResult.needsReindex) { @@ -1376,6 +1382,16 @@ export async function runFullAnalysis( await wipeLbugDbFiles(lbugPath); } + // Size the buffer pool to the graph just built by the pipeline (a page cache + // over the on-disk index, which scales with node/edge count) instead of the + // fixed 2 GiB default, whose eager commit dominates large-repo analyze. The + // size is clamped to [COPY-safety floor, default], so it only ever shrinks + // the pool; env override / no-hint paths are unchanged. See + // resolveBufferManagerSize / estimateBufferPool. + setBufferPoolSizeHint( + estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount), + ); + await initLbug(lbugPath); // Manual WAL checkpoint driver (#1741): periodically drain the WAL diff --git a/gitnexus/test/integration/skills-e2e.test.ts b/gitnexus/test/integration/skills-e2e.test.ts index 77ecc922d..581497880 100644 --- a/gitnexus/test/integration/skills-e2e.test.ts +++ b/gitnexus/test/integration/skills-e2e.test.ts @@ -2371,7 +2371,13 @@ export function createEntry(level: string, msg: string) { }); result1 = runSkillsCli(tmpDir); result2 = runSkillsCli(tmpDir); - }, 90000); + // 120s to match the other describe hooks in this file. This hook runs + // runSkillsCli TWICE, each capped at 45s, so a 90s budget has no headroom + // over two worst-case analyzes plus fixture setup and git init — it times + // out the *hook* on slow Windows runners (the test below already tolerates + // an individual analyze hitting its own 45s timeout via status === null, + // but a hook timeout fails before that tolerance can apply). + }, 120000); afterAll(() => { fs.rmSync(tmpDir, { recursive: true, force: true }); diff --git a/gitnexus/test/unit/http-embedder.test.ts b/gitnexus/test/unit/http-embedder.test.ts index 63cd715f9..fb4dde0af 100644 --- a/gitnexus/test/unit/http-embedder.test.ts +++ b/gitnexus/test/unit/http-embedder.test.ts @@ -9,6 +9,7 @@ const ENV_KEYS = [ 'GITNEXUS_EMBEDDING_MAX_ATTEMPTS', 'GITNEXUS_EMBEDDING_RETRY_CAP_MS', 'GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', + 'GITNEXUS_EMBEDDING_REQUEST_DIMS', ] as const; /** 384d mock vector matching the default schema dimensions. */ @@ -166,6 +167,30 @@ describe('HTTP embedding backend', () => { expect(result.length).toBe(1024); }); + it('can validate custom dims without forwarding dimensions to strict backends', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'omit'; + + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect('dimensions' in body).toBe(false); + expect(body.model).toBe('bge-m3'); + expect(result.length).toBe(1024); + }); + it('forwards dimensions on the single-query path', async () => { process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; process.env.GITNEXUS_EMBEDDING_MODEL = 'text-embedding-3-large'; @@ -188,6 +213,93 @@ describe('HTTP embedding backend', () => { expect(result.length).toBe(512); }); + it('can omit dimensions on the single-query path while validating custom dims', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'omit'; + + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const mod = await import('../../src/mcp/core/embedder.js'); + const result = await mod.embedQuery('query text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect('dimensions' in body).toBe(false); + expect(result.length).toBe(1024); + }); + + it.each(['none', 'off', 'false', '0'])( + 'treats GITNEXUS_EMBEDDING_REQUEST_DIMS=%s as omit and drops the request dimensions field', + async (alias) => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = alias; + + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect('dimensions' in body).toBe(false); + expect(result.length).toBe(1024); + }, + ); + + it('sends REQUEST_DIMS as the request dimensions while DIMS validates the response', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'text-embedding-3-large'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = '512'; + + // Response keeps the DIMS-validated length; only the outgoing request differs. + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect(body.dimensions).toBe(512); + expect(result.length).toBe(1024); + }); + + it('rejects a malformed GITNEXUS_EMBEDDING_REQUEST_DIMS with an error naming that var', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'garbage'; + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const { isHttpEmbeddingDimsError } = await import('../../src/core/embeddings/http-client.js'); + const err = await embedText('test').catch((e: unknown) => e); + // Recognizable as a config error so the CLI prints a clean message... + expect(isHttpEmbeddingDimsError(String(err))).toBe(true); + // ...and it points the operator at the var they set, not GITNEXUS_EMBEDDING_DIMS. + expect(String(err)).toContain('GITNEXUS_EMBEDDING_REQUEST_DIMS must be a positive integer'); + }); + it('retries on server error', async () => { process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; diff --git a/gitnexus/test/unit/lbug-config-wal.test.ts b/gitnexus/test/unit/lbug-config-wal.test.ts index a9912b9ba..94395541a 100644 --- a/gitnexus/test/unit/lbug-config-wal.test.ts +++ b/gitnexus/test/unit/lbug-config-wal.test.ts @@ -1,9 +1,11 @@ import os from 'os'; -import { describe, expect, it, vi } from 'vitest'; +import { afterEach, describe, expect, it, vi } from 'vitest'; import { createLbugDatabase, + estimateBufferPool, isLbugCheckpointIoError, isWalCorruptionError, + setBufferPoolSizeHint, } from '../../src/core/lbug/lbug-config.js'; import { _captureLogger } from '../../src/core/logger.js'; @@ -252,6 +254,74 @@ describe('createLbugDatabase buffer pool size (#2557)', () => { }); }); +describe('adaptive buffer pool hint', () => { + const GiB = 1024 * 1024 * 1024; + const MiB = 1024 * 1024; + const bufferPoolArg = (Database: ReturnType): unknown => Database.mock.calls[0][1]; + + afterEach(() => setBufferPoolSizeHint(undefined)); + + describe('estimateBufferPool', () => { + it.each([ + ['tiny graph clamps up to the 256 MiB COPY-safety floor', 41, 256 * MiB], + ['a graph under the floor still clamps up to 256 MiB', 40_000, 256 * MiB], + ['mid graph scales linearly (100k elements * 4 KiB = 400 MiB)', 100_000, 100_000 * 4 * 1024], + ['huge graph caps at the 2 GiB / 80%-RAM default', 10_000_000, 2 * GiB], + ])('%s', (_label, elements, expected) => { + const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB); + try { + expect(estimateBufferPool(elements)).toBe(expected); + } finally { + totalmemSpy.mockRestore(); + } + }); + }); + + it.each([ + ['a hint within range passes through', 512 * MiB, 512 * MiB], + ['a hint below the COPY-safety floor clamps up to 256 MiB', 100 * MiB, 256 * MiB], + ['a hint above the default clamps down to the 2 GiB cap', 8 * GiB, 2 * GiB], + ])( + 'createLbugDatabase uses the clamped hint when no env override is set: %s', + (_label, hint, expected) => { + const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB); + try { + setBufferPoolSizeHint(hint); + const Database = vi.fn(function (this: any) {}); + createLbugDatabase({ Database } as any, '/tmp/lbug-hint'); + expect(bufferPoolArg(Database)).toBe(expected); + } finally { + totalmemSpy.mockRestore(); + } + }, + ); + + it('env override wins over the hint (including 0 = native default)', () => { + try { + setBufferPoolSizeHint(128 * MiB); + vi.stubEnv('GITNEXUS_LBUG_BUFFER_POOL_SIZE', '0'); + const Database = vi.fn(function (this: any) {}); + createLbugDatabase({ Database } as any, '/tmp/lbug-hint-env'); + expect(bufferPoolArg(Database)).toBe(0); + } finally { + vi.unstubAllEnvs(); + } + }); + + it('falls back to the default when the hint is cleared', () => { + const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB); + try { + setBufferPoolSizeHint(128 * MiB); + setBufferPoolSizeHint(undefined); + const Database = vi.fn(function (this: any) {}); + createLbugDatabase({ Database } as any, '/tmp/lbug-hint-cleared'); + expect(bufferPoolArg(Database)).toBe(2 * GiB); + } finally { + totalmemSpy.mockRestore(); + } + }); +}); + // ─── Finding 8: strict + permissive checkpoint IO matchers ───────────────── describe('isLbugCheckpointIoError', () => { it.each([ diff --git a/gitnexus/test/unit/process-processor.test.ts b/gitnexus/test/unit/process-processor.test.ts index 18d5c7be3..09600c6de 100644 --- a/gitnexus/test/unit/process-processor.test.ts +++ b/gitnexus/test/unit/process-processor.test.ts @@ -3,6 +3,7 @@ import { processProcesses, type ProcessDetectionConfig, } from '../../src/core/ingestion/process-processor.js'; +import { computeDynamicMaxProcesses } from '../../src/core/ingestion/pipeline-phases/processes.js'; import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; import type { CommunityMembership } from '../../src/core/ingestion/community-processor.js'; @@ -522,4 +523,40 @@ describe('processProcesses', () => { expect(result.processes.length).toBeLessThanOrEqual(3); expect(result.stats.totalProcesses).toBeLessThanOrEqual(3); }); + + // Regression for #2198: the processesPhase dynamic sizing used to cap at + // Math.min(300, symbolCount/10). On large repos (>3000 symbols) that silently + // truncated the process index. The cap was removed by extracting + // computeDynamicMaxProcesses() — this test exercises the helper directly + // so it fails if someone reintroduces the 300 ceiling. + describe('computeDynamicMaxProcesses (#2198)', () => { + it('returns at least the floor of 20 for tiny repos', () => { + expect(computeDynamicMaxProcesses(0)).toBe(20); + expect(computeDynamicMaxProcesses(50)).toBe(20); // 50/10 = 5, floored to 20 + expect(computeDynamicMaxProcesses(199)).toBe(20); // 199/10 ≈ 20 + }); + + it('scales linearly within the old 0–3000 range', () => { + expect(computeDynamicMaxProcesses(500)).toBe(50); + expect(computeDynamicMaxProcesses(1000)).toBe(100); + expect(computeDynamicMaxProcesses(2999)).toBe(300); + }); + + it('grows past 300 for large repos — the regression that #2198 fixes', () => { + // 3001 symbols → 300 (just at the boundary) + expect(computeDynamicMaxProcesses(3001)).toBe(300); + // 3100 symbols → 310 — would have been capped to 300 before the fix + expect(computeDynamicMaxProcesses(3100)).toBe(310); + // 5000 symbols → 500 + expect(computeDynamicMaxProcesses(5000)).toBe(500); + // 28000 symbols (real-world large repo) → 2800 + expect(computeDynamicMaxProcesses(28000)).toBe(2800); + }); + + it('does NOT cap at 300 — fails if Math.min(300, ...) is reintroduced', () => { + const largeRepo = computeDynamicMaxProcesses(10000); + expect(largeRepo).toBe(1000); + expect(largeRepo).toBeGreaterThan(300); + }); + }); });