mirror of
https://github.com/zotero/zotero.git
synced 2026-10-09 03:18:01 +00:00
Read stored embedding hashes in one query per chunk
Indexing re-enqueues every eligible item on each start to find what changed, and checked each one's stored hash with its own query, so a large library ran thousands of queries at startup.
This commit is contained in:
parent
4896b759d5
commit
9ebbea584d
2 changed files with 65 additions and 9 deletions
|
|
@ -1099,23 +1099,42 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
} = {}) {
|
||||
await Zotero.Items.loadDataTypes(items, ['itemData']);
|
||||
|
||||
// Every start re-enqueues the whole library to find what changed, so
|
||||
// read the stored hashes in one query per chunk rather than one per
|
||||
// item
|
||||
let storedHashes = new Map();
|
||||
let itemIDs = items.map(item => item.id);
|
||||
let chunkSize = 500;
|
||||
for (let i = 0; i < itemIDs.length; i += chunkSize) {
|
||||
let chunk = itemIDs.slice(i, i + chunkSize);
|
||||
let rows = await Zotero.DB.queryAsync(
|
||||
"SELECT itemID, sourceHash FROM embeddings.itemEmbeddings WHERE itemID IN ("
|
||||
+ chunk.map(() => '?').join(',') + ")",
|
||||
chunk
|
||||
);
|
||||
for (let row of rows) {
|
||||
storedHashes.set(row.itemID, row.sourceHash);
|
||||
}
|
||||
}
|
||||
|
||||
let toEmbed = [];
|
||||
let toDelete = [];
|
||||
for (let item of items) {
|
||||
let text = _getItemText(item);
|
||||
if (!text) {
|
||||
await Zotero.DB.queryAsync(
|
||||
"DELETE FROM embeddings.itemEmbeddings WHERE itemID=?", item.id
|
||||
);
|
||||
if (storedHashes.has(item.id)) {
|
||||
toDelete.push(item.id);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
let hash = Zotero.Utilities.Internal.md5(text);
|
||||
let existing = await Zotero.DB.valueQueryAsync(
|
||||
"SELECT sourceHash FROM embeddings.itemEmbeddings WHERE itemID=?", item.id
|
||||
);
|
||||
if (existing !== hash) {
|
||||
if (storedHashes.get(item.id) !== hash) {
|
||||
toEmbed.push({ item, text, hash });
|
||||
}
|
||||
}
|
||||
if (toDelete.length) {
|
||||
await _deleteEmbeddings(toDelete);
|
||||
}
|
||||
|
||||
// Process one library at a time
|
||||
toEmbed.sort((a, b) => (a.item.libraryID - b.item.libraryID)
|
||||
|
|
|
|||
|
|
@ -196,9 +196,45 @@ describe("Zotero.Embeddings", function () {
|
|||
stub.restore();
|
||||
}
|
||||
});
|
||||
|
||||
it("should look up stored hashes without a query per item", async function () {
|
||||
this.timeout(60000);
|
||||
for (let i = 0; i < 5; i++) {
|
||||
await createDataObject('item', { title: "Batched lookup " + i });
|
||||
}
|
||||
|
||||
let vector = new Float32Array(4).fill(0.5);
|
||||
let embedStub = sinon.stub(Zotero.Embeddings, 'embedPassages')
|
||||
.callsFake(async texts => texts.map(() => vector));
|
||||
let stubs = [
|
||||
embedStub,
|
||||
sinon.stub(Zotero.Embeddings, 'isEnabled').returns(true),
|
||||
sinon.stub(Zotero.Embeddings, 'getModelVersion').returns('test-model/1'),
|
||||
sinon.stub(Zotero.Embeddings, 'isDownloaded').resolves(true),
|
||||
sinon.stub(Zotero.Embeddings, 'preloadModel').resolves()
|
||||
];
|
||||
let queries = [];
|
||||
let queryStub = sinon.stub(Zotero.DB, 'queryAsync')
|
||||
.callsFake(function (sql, ...rest) {
|
||||
queries.push(sql);
|
||||
return queryStub.wrappedMethod.call(this, sql, ...rest);
|
||||
});
|
||||
try {
|
||||
await Zotero.Embeddings.Indexing.startIndexing();
|
||||
}
|
||||
finally {
|
||||
queryStub.restore();
|
||||
stubs.forEach(stub => stub.restore());
|
||||
}
|
||||
|
||||
// The run has to have indexed something for this to mean anything
|
||||
assert.isTrue(embedStub.called);
|
||||
assert.isEmpty(queries.filter(sql => sql.includes('sourceHash')
|
||||
&& sql.includes('itemID=?')));
|
||||
assert.isNotEmpty(queries.filter(sql => sql.includes('itemID, sourceHash')));
|
||||
});
|
||||
});
|
||||
// Downloads the active model (~130 MB), so run explicitly:
|
||||
// ZOTERO_TEST_EMBEDDINGS_INFERENCE=1 test/runtests.sh -g "real vectors" embeddings
|
||||
|
||||
describe("#embed() with a real model", function () {
|
||||
before(function () {
|
||||
if (!Services.env.get("ZOTERO_TEST_EMBEDDINGS_INFERENCE")) {
|
||||
|
|
@ -244,6 +280,7 @@ describe("Zotero.Embeddings", function () {
|
|||
assert.isTrue(await Zotero.Embeddings.isDownloaded());
|
||||
});
|
||||
});
|
||||
|
||||
describe("memory pressure", function () {
|
||||
afterEach(function () {
|
||||
Services.obs.notifyObservers(null, 'memory-pressure-stop');
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue