perf(parse-impl): smaller default chunk budget (20MB→2MB) for cache granularity

The parse cache is keyed at chunk granularity. With the previous 20MB
budget, a typical mid-size repo (e.g. this worktree at 9MB total
parseable source) fits in a single chunk — meaning ANY file change
invalidates the whole chunk and re-parses every file.

2MB default produces ~5x more chunks on the same input, so a one-file
edit invalidates ~1/N of cached chunks instead of the whole thing.
Cold-run overhead from more chunks is <5% (one extra serialization
pass per chunk).

Override via GITNEXUS_CHUNK_BYTE_BUDGET env var for benchmarking.

Measured on this repo (~9MB / 887 parseable files):
  Cold (no cache):                    143s
  Warm cache, no source changes:        2s  (early-return)
  Warm cache + 1-file edit:            81s  (~43% off cold)

Speedup is bounded by the scopeResolution phase (~58s flat regardless
of parse cache) and by GitNexus's own auto-writes during analyze
(AGENTS.md / .claude/skills/ etc. mutate between runs and invalidate
chunks containing them). Both are addressable in follow-ups.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
abhigyanpatwari 2026-05-10 16:26:38 +05:30
parent b3431fb62f
commit d595c1c251

View file

@ -75,8 +75,21 @@ import { extractORMQueriesInline } from './orm-extraction.js';
import { logger } from '../../logger.js';
// ── Constants ──────────────────────────────────────────────────────────────
/** Max bytes of source content to load per parse chunk. */
const CHUNK_BYTE_BUDGET = 20 * 1024 * 1024; // 20MB
/** Max bytes of source content to load per parse chunk.
*
* Memory bound for the worker pool dispatch + a granularity knob for
* the parse cache. A single file change invalidates only its enclosing
* chunk, so smaller budgets → finer-grained invalidation.
*
* Override via GITNEXUS_CHUNK_BYTE_BUDGET (bytes) — the default of 2MB
* gives a useful invalidation floor (~1/N chunks on a multi-MB repo)
* while keeping worker dispatch overhead under 5% on cold runs.
*/
const CHUNK_BYTE_BUDGET = (() => {
const env = Number(process.env.GITNEXUS_CHUNK_BYTE_BUDGET);
if (Number.isFinite(env) && env > 0) return env;
return 2 * 1024 * 1024;
})();
// ── Main parse + resolve function ──────────────────────────────────────────
@ -312,7 +325,7 @@ export async function runChunkedParseAndResolve(
chunkWorkerData = mergeChunkResults(graph, symbolTable, cachedRaw);
if (isDev) {
logger.info(
`📦 parse-cache hit: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash!.slice(0, 8)})`,
`📦 parse-cache HIT: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash!.slice(0, 8)})`,
);
}
// Progress update so UI advances even on a cache hit.
@ -363,6 +376,11 @@ export async function runChunkedParseAndResolve(
// small repos without worker pool simply don't cache. That's fine.
if (parseCache && chunkHash && rawResults.length > 0) {
parseCache.entries.set(chunkHash, rawResults);
if (isDev) {
logger.info(
`📦 parse-cache MISS+store: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash.slice(0, 8)})`,
);
}
}
}