mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-03 02:21:44 +00:00
* fix(embeddings): spill cached vectors to a Float32 temp file Keep restore metadata in RAM and write embeddings once the in-memory row limit is exceeded so incremental analyze can survive large caches without a full-table number[] heap (#3306). Co-authored-by: Cursor <cursoragent@cursor.com> * fix(lbug): stream CodeEmbedding cache under the connection lock Spill vectors once the in-memory limit is crossed and fail the load instead of adopting an empty snapshot, so incremental analyze cannot OOM or quietly drop the restore cache (#3306). Co-authored-by: Cursor <cursoragent@cursor.com> * fix(analyze): restore cached embeddings from a streamed spill snapshot Hold row metadata across wipe, materialize 200-row batches, and treat cache-load failures as warn-and-continue so incremental analyze can preserve vectors without a full-table heap (#3306). Co-authored-by: Cursor <cursoragent@cursor.com> * Address PR review feedback (#3310) Loop spill writes until the full vector lands, keep materialize failures out of the insert catch and the Phase 4 hash skip-set, and assert spilled restore subsets by node id instead of scan order. Co-authored-by: Cursor <cursoragent@cursor.com> * Address PR review feedback (#3310) Discard only this analyze run's embedding spills so a concurrent analyze on another index keeps its restore file, and isolate the default in-memory limit test from inherited env. Co-authored-by: Cursor <cursoragent@cursor.com> * Address PR review feedback (#3310) Mark a node stale when any restore batch fails so leftover chunks are deleted and rembedded, and exercise a full-length bad-magic spill header. --------- Co-authored-by: Gergo Magyar <gergomagyar0@gmail.com> Co-authored-by: Cursor <cursoragent@cursor.com>
113 lines
4.6 KiB
TypeScript
113 lines
4.6 KiB
TypeScript
/**
|
|
* Real-DB coverage for #3306: loadCachedEmbeddings must stream CodeEmbedding
|
|
* rows instead of getAll()+map(Number) of the whole table, and must be able
|
|
* to spill vectors so incremental analyze does not keep every embedding in
|
|
* the V8 heap.
|
|
*/
|
|
import fs from 'node:fs';
|
|
import path from 'node:path';
|
|
import { afterEach, describe, expect, it } from 'vitest';
|
|
import { createTempDir, type TestDBHandle } from '../helpers/test-db.js';
|
|
import { EMBEDDING_DIMS } from '../../src/core/lbug/schema.js';
|
|
import { batchInsertEmbeddings } from '../../src/core/embeddings/embedding-pipeline.js';
|
|
import {
|
|
disposeEmbeddingSpill,
|
|
materializeCachedEmbeddings,
|
|
} from '../../src/core/embeddings/embedding-restore-spill.js';
|
|
|
|
describe('loadCachedEmbeddings streaming (#3306)', () => {
|
|
let tmp: TestDBHandle | undefined;
|
|
|
|
afterEach(async () => {
|
|
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
|
try {
|
|
await adapter.closeLbug();
|
|
} catch {
|
|
/* already closed */
|
|
}
|
|
await tmp?.cleanup();
|
|
tmp = undefined;
|
|
});
|
|
|
|
async function seedDb(rowCount: number) {
|
|
tmp = await createTempDir('gitnexus-lbug-');
|
|
const dbPath = path.join(tmp.dbPath, 'lbug');
|
|
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
|
await adapter.initLbug(dbPath);
|
|
const rows = Array.from({ length: rowCount }, (_, i) => ({
|
|
nodeId: `Function:src/f${i}.ts:fn${i}:1`,
|
|
chunkIndex: 0,
|
|
startLine: 1,
|
|
endLine: 3,
|
|
embedding: Array.from({ length: EMBEDDING_DIMS }, (__, d) => (d === 0 ? i + 1 : 0)),
|
|
contentHash: `hash-${i}`,
|
|
}));
|
|
await batchInsertEmbeddings(adapter.executeWithReusedStatement, rows);
|
|
return { adapter, rows };
|
|
}
|
|
|
|
it('materializes a small table in RAM (skip-fts / mock-compatible shape)', async () => {
|
|
const { adapter, rows } = await seedDb(3);
|
|
const cached = await adapter.loadCachedEmbeddings();
|
|
expect(cached.spill).toBeUndefined();
|
|
expect(cached.embeddings).toHaveLength(3);
|
|
expect(cached.rows).toHaveLength(3);
|
|
expect(cached.embeddingNodeIds.size).toBe(3);
|
|
expect(cached.embeddings.map((e) => e.nodeId).sort()).toEqual(rows.map((r) => r.nodeId).sort());
|
|
expect(cached.embeddings.find((e) => e.nodeId === rows[1]!.nodeId)?.embedding[0]).toBe(2);
|
|
});
|
|
|
|
it('streams into a spill file when the in-memory limit is 0 and restores a subset', async () => {
|
|
const { adapter, rows } = await seedDb(12);
|
|
const cached = await adapter.loadCachedEmbeddings({ inMemoryRowLimit: 0 });
|
|
try {
|
|
expect(cached.embeddings).toEqual([]);
|
|
expect(cached.spill?.rowCount).toBe(12);
|
|
expect(cached.rows).toHaveLength(12);
|
|
const wanted = new Set([rows[0]!.nodeId, rows[5]!.nodeId, rows[10]!.nodeId]);
|
|
const subset = materializeCachedEmbeddings(
|
|
cached,
|
|
cached.rows.filter((meta) => wanted.has(meta.nodeId)),
|
|
);
|
|
expect(subset).toHaveLength(3);
|
|
const byId = new Map(subset.map((row) => [row.nodeId, row]));
|
|
expect(byId.get(rows[0]!.nodeId)?.embedding[0]).toBe(1);
|
|
expect(byId.get(rows[5]!.nodeId)?.embedding[0]).toBe(6);
|
|
expect(byId.get(rows[10]!.nodeId)?.embedding[0]).toBe(11);
|
|
expect(byId.get(rows[0]!.nodeId)?.contentHash).toBe(rows[0]!.contentHash);
|
|
} finally {
|
|
disposeEmbeddingSpill(cached.spill);
|
|
}
|
|
});
|
|
|
|
it('flips from RAM to spill once a non-zero in-memory limit is crossed', async () => {
|
|
const { adapter, rows } = await seedDb(8);
|
|
const cached = await adapter.loadCachedEmbeddings({ inMemoryRowLimit: 4 });
|
|
try {
|
|
expect(cached.embeddings).toEqual([]);
|
|
expect(cached.spill?.rowCount).toBe(8);
|
|
expect(cached.rows).toHaveLength(8);
|
|
expect(fs.statSync(cached.spill!.path).size).toBe(12 + 8 * EMBEDDING_DIMS * 4);
|
|
const wanted = new Set([rows[2]!.nodeId, rows[3]!.nodeId]);
|
|
const subset = materializeCachedEmbeddings(
|
|
cached,
|
|
cached.rows.filter((meta) => wanted.has(meta.nodeId)),
|
|
);
|
|
expect(subset).toHaveLength(2);
|
|
const byId = new Map(subset.map((row) => [row.nodeId, row]));
|
|
expect(byId.get(rows[2]!.nodeId)?.embedding[0]).toBe(3);
|
|
expect(byId.get(rows[3]!.nodeId)?.embedding[0]).toBe(4);
|
|
} finally {
|
|
disposeEmbeddingSpill(cached.spill);
|
|
}
|
|
});
|
|
|
|
it('surfaces a spill write failure instead of adopting an empty snapshot', async () => {
|
|
const { adapter } = await seedDb(3);
|
|
const spillDir = path.join(tmp!.dbPath, 'not-a-directory');
|
|
fs.writeFileSync(spillDir, 'x');
|
|
await expect(adapter.loadCachedEmbeddings({ inMemoryRowLimit: 0, spillDir })).rejects.toThrow(
|
|
/ENOTDIR|not a directory|ENOSPC|EACCES/i,
|
|
);
|
|
});
|
|
});
|