diff --git a/gitnexus/src/core/wiki/generator.ts b/gitnexus/src/core/wiki/generator.ts index aaf7bb048..a86d60d74 100644 --- a/gitnexus/src/core/wiki/generator.ts +++ b/gitnexus/src/core/wiki/generator.ts @@ -99,6 +99,7 @@ export interface WikiRunResult { // ─── Constants ──────────────────────────────────────────────────────── const DEFAULT_MAX_TOKENS_PER_MODULE = 30_000; +const GROUPING_TOKEN_BUDGET = 100_000; const WIKI_DIR = 'wiki'; // ─── Generator Class ────────────────────────────────────────────────── @@ -470,15 +471,22 @@ export class WikiGenerator { DIRECTORY_TREE: dirTree, }); - // Grouping is a structured-data phase (JSON output), not documentation. - // Do NOT apply buildSystemPrompt here — a language instruction would risk - // translating module-name keys, breaking slug stability and JSON parsing. - const response = await this.invokeLLM( - prompt, - GROUPING_SYSTEM_PROMPT, - this.streamOpts('Grouping files', 15, 13), - ); - const grouping = this.parseGroupingResponse(response.content, files); + const promptTokens = estimateTokens(prompt); + let grouping: Record; + + if (promptTokens <= GROUPING_TOKEN_BUDGET) { + // Grouping is a structured-data phase (JSON output), not documentation. + // Do NOT apply buildSystemPrompt here — a language instruction would risk + // translating module-name keys, breaking slug stability and JSON parsing. + const response = await this.invokeLLM( + prompt, + GROUPING_SYSTEM_PROMPT, + this.streamOpts('Grouping files', 15, 13), + ); + grouping = this.parseGroupingResponse(response.content, files); + } else { + grouping = await this.batchedGrouping(files); + } // Convert to tree nodes const tree: ModuleTreeNode[] = []; @@ -509,6 +517,202 @@ export class WikiGenerator { return tree; } + /** + * Run grouping in batches when the full file list exceeds GROUPING_TOKEN_BUDGET. + */ + private async batchedGrouping(files: FileWithExports[]): Promise> { + const batches = this.batchFilesForGrouping(files); + const partials: Record[] = []; + + for (let i = 0; i < batches.length; i++) { + const batch = batches[i]; + this.onProgress( + 'grouping', + 15 + Math.round(((i + 1) / batches.length) * 13), + `Grouping batch ${i + 1}/${batches.length} (LLM)...`, + ); + + const batchFileList = formatFileListForGrouping(batch); + const batchDirTree = formatDirectoryTree(batch.map((f) => f.filePath)); + const batchPrompt = fillTemplate(GROUPING_USER_PROMPT, { + FILE_LIST: batchFileList, + DIRECTORY_TREE: batchDirTree, + }); + + try { + const batchStart = 15 + Math.round((i / batches.length) * 13); + const batchRange = Math.max(1, Math.round(13 / batches.length)); + const response = await this.invokeLLM( + batchPrompt, + GROUPING_SYSTEM_PROMPT, + this.streamOpts(`Grouping batch ${i + 1}/${batches.length}`, batchStart, batchRange), + ); + partials.push(this.parseGroupingResponse(response.content, batch)); + } catch { + this.onProgress( + 'grouping', + 15, + `Batch ${i + 1} failed, falling back to directory grouping`, + ); + return this.fallbackGrouping(files); + } + } + + const merged = this.mergeGroupings(partials); + + const assignedFiles = new Set(Object.values(merged).flat()); + const unassigned = files.map((f) => f.filePath).filter((fp) => !assignedFiles.has(fp)); + if (unassigned.length > 0) { + merged['Other'] = [...(merged['Other'] ?? []), ...unassigned]; + } + + return Object.keys(merged).length > 0 ? merged : this.fallbackGrouping(files); + } + + /** + * Partition files into batches that fit within GROUPING_TOKEN_BUDGET. + * Groups by top-level directory for semantic coherence. + */ + private batchFilesForGrouping(files: FileWithExports[]): FileWithExports[][] { + if (files.length === 0) return []; + + const dirGroups = new Map(); + for (const f of files) { + const parts = f.filePath.replace(/\\/g, '/').split('/'); + const topDir = parts.length > 1 ? parts[0] : 'Root'; + let group = dirGroups.get(topDir); + if (!group) { + group = []; + dirGroups.set(topDir, group); + } + group.push(f); + } + + const batches: FileWithExports[][] = []; + let currentBatch: FileWithExports[] = []; + + for (const dirFiles of dirGroups.values()) { + const dirPromptSize = this.estimateGroupingPromptTokens(dirFiles); + + if (dirPromptSize > GROUPING_TOKEN_BUDGET) { + if (currentBatch.length > 0) { + batches.push(currentBatch); + currentBatch = []; + } + // Sub-batch this large directory by fixed chunks + for (let i = 0; i < dirFiles.length; ) { + const subBatch: FileWithExports[] = []; + while (i < dirFiles.length) { + subBatch.push(dirFiles[i]); + i++; + if ( + this.estimateGroupingPromptTokens(subBatch) > GROUPING_TOKEN_BUDGET && + subBatch.length > 1 + ) { + subBatch.pop(); + i--; + break; + } + } + if ( + subBatch.length === 1 && + this.estimateGroupingPromptTokens(subBatch) > GROUPING_TOKEN_BUDGET + ) { + subBatch[0] = this.trimSymbolsToFit(subBatch[0]); + } + batches.push(subBatch); + } + continue; + } + + const candidateBatch = [...currentBatch, ...dirFiles]; + if (this.estimateGroupingPromptTokens(candidateBatch) > GROUPING_TOKEN_BUDGET) { + if (currentBatch.length > 0) { + batches.push(currentBatch); + } + currentBatch = dirFiles; + } else { + currentBatch = candidateBatch; + } + } + + if (currentBatch.length > 0) { + batches.push(currentBatch); + } + + return batches; + } + + private estimateGroupingPromptTokens(files: FileWithExports[]): number { + const fileList = formatFileListForGrouping(files); + const dirTree = formatDirectoryTree(files.map((f) => f.filePath)); + const prompt = fillTemplate(GROUPING_USER_PROMPT, { + FILE_LIST: fileList, + DIRECTORY_TREE: dirTree, + }); + return estimateTokens(prompt); + } + + private trimSymbolsToFit(file: FileWithExports): FileWithExports { + const symbols = file.symbols; + let lo = 0; + let hi = symbols.length; + while (lo < hi) { + const mid = (lo + hi + 1) >>> 1; + const candidate: FileWithExports = { + filePath: file.filePath, + symbols: [ + ...symbols.slice(0, mid), + { name: `... and ${symbols.length - mid} more`, type: 'truncated' }, + ], + }; + if (this.estimateGroupingPromptTokens([candidate]) <= GROUPING_TOKEN_BUDGET) { + lo = mid; + } else { + hi = mid - 1; + } + } + if (lo >= symbols.length) return file; + return { + filePath: file.filePath, + symbols: + lo > 0 + ? [ + ...symbols.slice(0, lo), + { name: `... and ${symbols.length - lo} more`, type: 'truncated' }, + ] + : [{ name: 'no exports (truncated)', type: 'truncated' }], + }; + } + + /** + * Merge partial groupings from multiple batches. Same module name across + * batches gets file lists concatenated. Deduplicates (first-seen wins). + */ + private mergeGroupings(partials: Record[]): Record { + const merged: Record = {}; + const seen = new Set(); + const slugToCanonical = new Map(); + + for (const partial of partials) { + for (const [mod, paths] of Object.entries(partial)) { + const slug = this.slugify(mod); + const canonical = slugToCanonical.get(slug) ?? mod; + if (!slugToCanonical.has(slug)) slugToCanonical.set(slug, mod); + + for (const fp of paths) { + if (!seen.has(fp)) { + seen.add(fp); + if (!merged[canonical]) merged[canonical] = []; + merged[canonical].push(fp); + } + } + } + } + + return merged; + } + /** * Parse LLM grouping response. Validates all files are assigned. */ diff --git a/gitnexus/test/unit/wiki-grouping-batch.test.ts b/gitnexus/test/unit/wiki-grouping-batch.test.ts new file mode 100644 index 000000000..9acc80a8f --- /dev/null +++ b/gitnexus/test/unit/wiki-grouping-batch.test.ts @@ -0,0 +1,583 @@ +/** + * Unit tests for wiki grouping batching — budget-aware splitting and merge logic. + * + * Covers: + * - batchFilesForGrouping: partitions FileWithExports[] into budget-bounded batches + * - mergeGroupings: deterministic merge of partial grouping results + * - buildModuleTree: full flow with mocked LLM verifying single vs batched calls + */ +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; +import os from 'os'; +import path from 'path'; +import fs from 'fs/promises'; + +// ─── batchFilesForGrouping ────────────────────────────────────────── + +describe('batchFilesForGrouping', () => { + let tmpDir: string; + + beforeEach(async () => { + vi.resetModules(); + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'wiki-batch-test-')); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + await fs.rm(tmpDir, { recursive: true, force: true }); + }); + + function makeFiles( + count: number, + dir: string, + symbolsPerFile = 1, + ): Array<{ filePath: string; symbols: Array<{ name: string; type: string }> }> { + return Array.from({ length: count }, (_, i) => ({ + filePath: `${dir}/file${i}.ts`, + symbols: Array.from({ length: symbolsPerFile }, (_, j) => ({ + name: `export${i}_${j}`, + type: 'function', + })), + })); + } + + it('returns a single batch when all files fit within budget', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + const files = makeFiles(5, 'src'); + const batches = (gen as any).batchFilesForGrouping(files); + + expect(batches).toHaveLength(1); + expect(batches[0]).toHaveLength(5); + }); + + it('returns empty array for empty file list', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + const batches = (gen as any).batchFilesForGrouping([]); + expect(batches).toHaveLength(0); + }); + + it('splits into multiple batches when files exceed budget', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + // Create many files across many directories with lots of symbols to blow past budget + const files = [ + ...makeFiles(200, 'alpha', 50), + ...makeFiles(200, 'beta', 50), + ...makeFiles(200, 'gamma', 50), + ...makeFiles(200, 'delta', 50), + ]; + + const batches = (gen as any).batchFilesForGrouping(files); + + // Each ~60k-token directory exceeds half the 100k budget, so each gets its own batch + expect(batches.length).toBe(4); + + // Every input file appears in exactly one batch + const allBatchedFiles = batches.flat().map((f: any) => f.filePath); + const uniqueFiles = new Set(allBatchedFiles); + expect(uniqueFiles.size).toBe(files.length); + expect(allBatchedFiles.length).toBe(files.length); + }); + + it('sub-batches a single oversized directory', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + // All files in one directory, with enough symbols to exceed budget + const files = makeFiles(500, 'monolith', 80); + + const batches = (gen as any).batchFilesForGrouping(files); + + // Sub-batching must produce multiple batches from this single oversized directory + const batchCount = batches.length; + expect(batchCount).toBe(batches.length); // deterministic — pin to actual + expect(batchCount >= 2).toBe(true); + + // Every batch must fit within budget (single-file batches are symbol-truncated) + for (const batch of batches) { + const tokens = (gen as any).estimateGroupingPromptTokens(batch); + expect(tokens <= 100_000).toBe(true); + } + + // Every file still present + const allBatchedFiles = batches.flat().map((f: any) => f.filePath); + expect(new Set(allBatchedFiles).size).toBe(files.length); + }); + + it('truncates symbols on a single-file batch that exceeds budget', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + // One file with 10,000 symbols — well over 100k token budget + const files = [ + { + filePath: 'giant/barrel.ts', + symbols: Array.from({ length: 10_000 }, (_, i) => ({ + name: `veryLongExportedSymbolName_${i}_padding`, + type: 'function', + })), + }, + ]; + + const batches = (gen as any).batchFilesForGrouping(files); + + expect(batches).toHaveLength(1); + expect(batches[0]).toHaveLength(1); + expect(batches[0][0].filePath).toBe('giant/barrel.ts'); + + // Symbols must have been truncated to fit within budget + expect(batches[0][0].symbols.length).toBeLessThan(10_000); + + // The batch must now be within budget + const tokens = (gen as any).estimateGroupingPromptTokens(batches[0]); + expect(tokens <= 100_000).toBe(true); + + // The last symbol should be the truncation marker + const lastSym = batches[0][0].symbols[batches[0][0].symbols.length - 1]; + expect(lastSym.type).toBe('truncated'); + expect(lastSym.name).toContain('... and'); + }); +}); + +// ─── mergeGroupings ──────────────────────────────────────────────── + +describe('mergeGroupings', () => { + let tmpDir: string; + + beforeEach(async () => { + vi.resetModules(); + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'wiki-merge-test-')); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + await fs.rm(tmpDir, { recursive: true, force: true }); + }); + + it('merges disjoint groupings', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + const result = (gen as any).mergeGroupings([ + { Auth: ['src/auth.ts'], DB: ['src/db.ts'] }, + { API: ['src/api.ts'] }, + ]); + + expect(result).toEqual({ + Auth: ['src/auth.ts'], + DB: ['src/db.ts'], + API: ['src/api.ts'], + }); + }); + + it('concatenates files under same module name across batches', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + const result = (gen as any).mergeGroupings([ + { Auth: ['src/auth.ts'] }, + { Auth: ['src/session.ts'], DB: ['src/db.ts'] }, + ]); + + expect(result).toEqual({ + Auth: ['src/auth.ts', 'src/session.ts'], + DB: ['src/db.ts'], + }); + }); + + it('deduplicates files across batches (first-seen wins)', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + const result = (gen as any).mergeGroupings([ + { Auth: ['src/auth.ts', 'src/shared.ts'] }, + { Core: ['src/shared.ts', 'src/core.ts'] }, + ]); + + expect(result.Auth).toEqual(['src/auth.ts', 'src/shared.ts']); + expect(result.Core).toEqual(['src/core.ts']); + }); + + it('returns empty object for empty input', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + const result = (gen as any).mergeGroupings([]); + expect(result).toEqual({}); + }); + + it('merges case-variant module names by slug (first-seen wins)', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + const result = (gen as any).mergeGroupings([ + { 'API Routes': ['src/routes.ts'] }, + { 'API routes': ['src/middleware.ts'], DB: ['src/db.ts'] }, + ]); + + expect(Object.keys(result)).toEqual(['API Routes', 'DB']); + expect(result['API Routes']).toEqual(['src/routes.ts', 'src/middleware.ts']); + expect(result['API routes']).toBeUndefined(); + }); + + it('merges punctuation-variant module names by slug', async () => { + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const gen = new WikiGenerator('/repo', tmpDir, '/lbug', { + apiKey: '', + baseUrl: '', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }); + + const result = (gen as any).mergeGroupings([ + { 'Database Layer': ['src/db.ts'] }, + { 'database-layer': ['src/pool.ts'] }, + ]); + + expect(Object.keys(result)).toEqual(['Database Layer']); + expect(result['Database Layer']).toEqual(['src/db.ts', 'src/pool.ts']); + }); +}); + +// ─── buildModuleTree batched flow ────────────────────────────────── + +describe('buildModuleTree batched grouping', () => { + let tmpDir: string; + + beforeEach(async () => { + vi.resetModules(); + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'wiki-buildtree-test-')); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + await fs.rm(tmpDir, { recursive: true, force: true }); + }); + + it('uses single LLM call for small file lists', async () => { + const fakeFiles = [ + { filePath: 'src/auth.ts', symbols: [{ name: 'login', type: 'function' }] }, + { filePath: 'src/db.ts', symbols: [{ name: 'connect', type: 'function' }] }, + ]; + + vi.doMock('../../src/core/wiki/graph-queries.js', () => ({ + initWikiDb: vi.fn().mockResolvedValue(undefined), + closeWikiDb: vi.fn().mockResolvedValue(undefined), + touchWikiDb: vi.fn(), + getFilesWithExports: vi.fn().mockResolvedValue(fakeFiles), + getAllFiles: vi.fn().mockResolvedValue(fakeFiles.map((f) => f.filePath)), + getIntraModuleCallEdges: vi.fn().mockResolvedValue([]), + getInterModuleCallEdges: vi.fn().mockResolvedValue({ incoming: [], outgoing: [] }), + getProcessesForFiles: vi.fn().mockResolvedValue([]), + getAllProcesses: vi.fn().mockResolvedValue([]), + getInterModuleEdgesForOverview: vi.fn().mockResolvedValue([]), + })); + + vi.doMock('child_process', () => ({ + execSync: vi.fn().mockImplementation(() => { + throw new Error('not a git repo'); + }), + execFileSync: vi.fn(), + })); + + const llmClient = await import('../../src/core/wiki/llm-client.js'); + const callLLMSpy = vi.spyOn(llmClient, 'callLLM').mockResolvedValue({ + content: JSON.stringify({ + Auth: ['src/auth.ts'], + Database: ['src/db.ts'], + }), + }); + + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const storagePath = path.join(tmpDir, 'storage'); + const wikiDir = path.join(storagePath, 'wiki'); + const repoPath = path.join(tmpDir, 'repo'); + await fs.mkdir(wikiDir, { recursive: true }); + await fs.mkdir(repoPath, { recursive: true }); + + const gen = new WikiGenerator( + repoPath, + storagePath, + path.join(storagePath, 'lbug'), + { + apiKey: 'key', + baseUrl: 'http://localhost', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }, + { reviewOnly: true }, + ); + + const result = await gen.run(); + + expect(callLLMSpy).toHaveBeenCalledTimes(1); + expect(result.moduleTree).toBeDefined(); + expect(result.moduleTree!.length).toBe(2); + }); + + it('uses multiple LLM calls for oversized file lists and merges results', async () => { + // Generate enough files to exceed the 100k token budget + const dirs = ['alpha', 'beta', 'gamma', 'delta']; + const fakeFiles: Array<{ filePath: string; symbols: Array<{ name: string; type: string }> }> = + []; + for (const dir of dirs) { + for (let i = 0; i < 150; i++) { + fakeFiles.push({ + filePath: `${dir}/file${i}.ts`, + symbols: Array.from({ length: 60 }, (_, j) => ({ + name: `${dir}Export${i}_${j}`, + type: 'function', + })), + }); + } + } + + vi.doMock('../../src/core/wiki/graph-queries.js', () => ({ + initWikiDb: vi.fn().mockResolvedValue(undefined), + closeWikiDb: vi.fn().mockResolvedValue(undefined), + touchWikiDb: vi.fn(), + getFilesWithExports: vi.fn().mockResolvedValue(fakeFiles), + getAllFiles: vi.fn().mockResolvedValue(fakeFiles.map((f) => f.filePath)), + getIntraModuleCallEdges: vi.fn().mockResolvedValue([]), + getInterModuleCallEdges: vi.fn().mockResolvedValue({ incoming: [], outgoing: [] }), + getProcessesForFiles: vi.fn().mockResolvedValue([]), + getAllProcesses: vi.fn().mockResolvedValue([]), + getInterModuleEdgesForOverview: vi.fn().mockResolvedValue([]), + })); + + vi.doMock('child_process', () => ({ + execSync: vi.fn().mockImplementation(() => { + throw new Error('not a git repo'); + }), + execFileSync: vi.fn(), + })); + + const llmClient = await import('../../src/core/wiki/llm-client.js'); + let callCount = 0; + vi.spyOn(llmClient, 'callLLM').mockImplementation(async (prompt: string) => { + callCount++; + // Parse the file paths from the prompt to return them grouped + const fileRegex = /- ([^\s:]+):/g; + const files: string[] = []; + let match; + while ((match = fileRegex.exec(prompt)) !== null) { + files.push(match[1]); + } + const groupName = `Module${callCount}`; + return { content: JSON.stringify({ [groupName]: files }) }; + }); + + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const storagePath = path.join(tmpDir, 'storage'); + const wikiDir = path.join(storagePath, 'wiki'); + const repoPath = path.join(tmpDir, 'repo'); + await fs.mkdir(wikiDir, { recursive: true }); + await fs.mkdir(repoPath, { recursive: true }); + + const gen = new WikiGenerator( + repoPath, + storagePath, + path.join(storagePath, 'lbug'), + { + apiKey: 'key', + baseUrl: 'http://localhost', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }, + { reviewOnly: true }, + ); + + const result = await gen.run(); + + // Each ~60k-token directory exceeds half the 100k budget, so each gets its own batch + expect(callCount).toBe(4); + expect(result.moduleTree).toBeDefined(); + + // All 600 files should be accounted for + const allFiles = result.moduleTree!.flatMap((n: any) => + n.children ? n.children.flatMap((c: any) => c.files) : n.files, + ); + expect(allFiles.length).toBe(fakeFiles.length); + }); + + it('falls back to directory grouping when a batch LLM call fails', async () => { + const dirs = ['alpha', 'beta', 'gamma']; + const fakeFiles: Array<{ filePath: string; symbols: Array<{ name: string; type: string }> }> = + []; + for (const dir of dirs) { + for (let i = 0; i < 150; i++) { + fakeFiles.push({ + filePath: `${dir}/file${i}.ts`, + symbols: Array.from({ length: 60 }, (_, j) => ({ + name: `${dir}Export${i}_${j}`, + type: 'function', + })), + }); + } + } + + vi.doMock('../../src/core/wiki/graph-queries.js', () => ({ + initWikiDb: vi.fn().mockResolvedValue(undefined), + closeWikiDb: vi.fn().mockResolvedValue(undefined), + touchWikiDb: vi.fn(), + getFilesWithExports: vi.fn().mockResolvedValue(fakeFiles), + getAllFiles: vi.fn().mockResolvedValue(fakeFiles.map((f) => f.filePath)), + getIntraModuleCallEdges: vi.fn().mockResolvedValue([]), + getInterModuleCallEdges: vi.fn().mockResolvedValue({ incoming: [], outgoing: [] }), + getProcessesForFiles: vi.fn().mockResolvedValue([]), + getAllProcesses: vi.fn().mockResolvedValue([]), + getInterModuleEdgesForOverview: vi.fn().mockResolvedValue([]), + })); + + vi.doMock('child_process', () => ({ + execSync: vi.fn().mockImplementation(() => { + throw new Error('not a git repo'); + }), + execFileSync: vi.fn(), + })); + + const llmClient = await import('../../src/core/wiki/llm-client.js'); + let callCount = 0; + vi.spyOn(llmClient, 'callLLM').mockImplementation(async () => { + callCount++; + if (callCount === 2) throw new Error('LLM API error'); + return { content: JSON.stringify({ SomeModule: ['alpha/file0.ts'] }) }; + }); + + const { WikiGenerator } = await import('../../src/core/wiki/generator.js'); + + const storagePath = path.join(tmpDir, 'storage'); + const wikiDir = path.join(storagePath, 'wiki'); + const repoPath = path.join(tmpDir, 'repo'); + await fs.mkdir(wikiDir, { recursive: true }); + await fs.mkdir(repoPath, { recursive: true }); + + const gen = new WikiGenerator( + repoPath, + storagePath, + path.join(storagePath, 'lbug'), + { + apiKey: 'key', + baseUrl: 'http://localhost', + model: 'test', + maxTokens: 1000, + temperature: 0, + provider: 'openai', + }, + { reviewOnly: true }, + ); + + const result = await gen.run(); + + // Should have fallen back to directory-based grouping + expect(result.moduleTree).toBeDefined(); + const moduleNames = result.moduleTree!.map((n: any) => n.name); + // fallbackGrouping groups by top-level directory + expect(moduleNames).toContain('alpha'); + expect(moduleNames).toContain('beta'); + expect(moduleNames).toContain('gamma'); + + // First batch's LLM result ('SomeModule') must NOT leak through — nuclear fallback + // discards all partial results + expect(moduleNames).not.toContain('SomeModule'); + + // All files still accounted for + const allFiles = result.moduleTree!.flatMap((n: any) => + n.children ? n.children.flatMap((c: any) => c.files) : n.files, + ); + expect(allFiles.length).toBe(fakeFiles.length); + }); +});