diff --git a/chrome/content/zotero/preferences/preferences_advanced.js b/chrome/content/zotero/preferences/preferences_advanced.js
index dc937268e2..a659c16f8d 100644
--- a/chrome/content/zotero/preferences/preferences_advanced.js
+++ b/chrome/content/zotero/preferences/preferences_advanced.js
@@ -231,6 +231,69 @@ Zotero_Preferences.Advanced = {
document.getElementById('semantic-search-attachments-row').hidden
= !Zotero.Prefs.get('embeddings.indexFulltext');
this._updateSemanticSearchBar('attachments', status.chunks);
+ this._updateSemanticSearchDiagnostics(status.diagnostics, status.eta);
+ },
+
+
+ // Key/value rows of pipeline diagnostics for developers, so the labels
+ // are plain English rather than localized
+ _updateSemanticSearchDiagnostics: function (diagnostics, eta) {
+ let n = (value, digits = 0) => (value ?? 0).toLocaleString(undefined, {
+ maximumFractionDigits: digits, minimumFractionDigits: digits
+ });
+ let pct = value => Math.round((value || 0) * 100) + '%';
+ let mb = bytes => n(bytes / 1024 / 1024) + ' MB';
+ let speed = rate => `${n(rate.chunksPerSecond, 1)} chunks/s, ${n(rate.tokensPerSecond)} tokens/s`;
+ let bucketList = (buckets, total, unit) => buckets.map(({ from, to, count }) => {
+ let range = from === null ? `< ${n(to)}` : (to === null ? `≥ ${n(from)}` : `${n(from)}–${n(to - 1)}`);
+ return `${range}${unit}: ${n(count)} (${pct(total ? count / total : 0)})`;
+ }).join(' · ');
+ let proc = p => (p ? `${mb(p.memory)}${p.cpu === null ? '' : `, CPU ${p.cpu}%`}` : '—');
+ let duration = (seconds) => {
+ let h = Math.floor(seconds / 3600);
+ let m = Math.floor(seconds % 3600 / 60);
+ return h ? `${h} h ${m} m` : `${m} m ${Math.floor(seconds % 60)} s`;
+ };
+
+ let rows = [];
+ let { window, run, engine, processes, slice, chunks } = diagnostics;
+ rows.push(['ETA', eta === null ? '—' : duration(eta)]);
+ rows.push(['Throughput (2 min)', window ? speed(window) : '—']);
+ rows.push(['Inference speed (run)', run ? speed(run) : '—']);
+ rows.push(['Padding efficiency', window || run
+ ? `${window ? pct(window.paddingEfficiency) : '—'} (2 min), ${run ? pct(run.paddingEfficiency) : '—'} (run)`
+ : '—']);
+ rows.push(['Batches (run)', run
+ ? `${n(run.batches)} · ${n(run.chunksPerBatch, 1)} chunks · ${n(run.tokensPerBatch)} tokens avg`
+ : '—']);
+ rows.push(['Token budget', `${n(diagnostics.tokenBudget)} · ${n(diagnostics.pressureEvents)} memory-pressure events`]);
+ rows.push(['Engine threads', `${engine.threads} of ${engine.optimalThreads}`
+ + (engine.boosts.length ? ` (boost: ${engine.boosts.join(', ')})` : '')]);
+ rows.push(['Engine restarts (run)', `memory ${diagnostics.restarts.memory}, threads ${diagnostics.restarts.threads}`]);
+ rows.push(['Inference process', proc(processes?.inference)]);
+ rows.push(['Main process', proc(processes?.main)]);
+ rows.push(['Available memory', processes?.available ? mb(processes.available) : '—']);
+ rows.push(['Slice', slice ? `${n(slice.done)} / ${n(slice.total)} chunks` : '—']);
+ if (chunks) {
+ let { sizes, perDocument } = chunks;
+ rows.push(['Chunks stored', `${n(sizes.count)} · ${n(sizes.tokens)} tokens · mean ${n(sizes.mean)} · median ${n(sizes.median)}`]);
+ rows.push(['Chunk sizes', bucketList(sizes.buckets, sizes.count, '')]);
+ rows.push(['Split sections', `${pct(sizes.splitShare)} of chunks · ${n(sizes.partsPerSplitSection, 1)} parts per split section`]);
+ rows.push(['Chunks per document', `${n(perDocument.count)} documents · mean ${n(perDocument.mean, 1)} · median ${n(perDocument.median)} · max ${n(perDocument.max)}`]);
+ rows.push(['Documents by chunks', bucketList(perDocument.buckets, perDocument.count, '')]);
+ }
+
+ let box = document.getElementById('semantic-search-diagnostics');
+ while (box.childElementCount < rows.length * 2) {
+ box.append(document.createXULElement('label'), document.createXULElement('label'));
+ }
+ while (box.childElementCount > rows.length * 2) {
+ box.lastElementChild.remove();
+ }
+ rows.forEach(([label, value], i) => {
+ box.children[i * 2].value = label;
+ box.children[i * 2 + 1].value = value;
+ });
},
diff --git a/chrome/content/zotero/preferences/preferences_advanced.xhtml b/chrome/content/zotero/preferences/preferences_advanced.xhtml
index 0eaa19b0de..fb2bc89d49 100644
--- a/chrome/content/zotero/preferences/preferences_advanced.xhtml
+++ b/chrome/content/zotero/preferences/preferences_advanced.xhtml
@@ -355,6 +355,7 @@
+
diff --git a/chrome/content/zotero/xpcom/embeddings.js b/chrome/content/zotero/xpcom/embeddings.js
index d21f6433b2..49c3ffcef2 100644
--- a/chrome/content/zotero/xpcom/embeddings.js
+++ b/chrome/content/zotero/xpcom/embeddings.js
@@ -287,7 +287,7 @@ Zotero.Embeddings = new function () {
// Schema version of the attached embeddings database. The tables are only
// created when this is bumped (_setUpDB() drops and recreates everything),
// so any schema change needs a bump.
- const _dbVersion = 4;
+ const _dbVersion = 5;
let _dbInitPromise = null;
let _dbHooksRegistered = false;
@@ -391,9 +391,16 @@ Zotero.Embeddings = new function () {
+ " textCheck TEXT,\n"
+ " sectionPart INTEGER,\n"
+ " sectionParts INTEGER,\n"
+ + " tokens INTEGER,\n"
+ " PRIMARY KEY (itemID, chunkIndex)\n"
+ ")"
);
+ // Chunk-shape diagnostics read this index alone, never the rows with
+ // their vectors (see Indexing._getChunkShape())
+ await Zotero.DB.queryAsync(
+ "CREATE INDEX embeddings.itemEmbeddings_tokens "
+ + "ON itemEmbeddings (tokens, sectionParts, sectionPart)"
+ );
// The localUserKey the vectors were built against and the model that
// produced them
await Zotero.DB.queryAsync(
@@ -1688,71 +1695,9 @@ Zotero.Embeddings.Indexing = new function () {
// gets. Restarting the engine is the only reclaim and costs about a
// second, so past this footprint it's restarted between batches.
const INFERENCE_MEMORY_CAP = 1.5 * 1024 * 1024 * 1024;
- // How often process usage is sampled and logged during a run
- const PROC_SAMPLE_INTERVAL = 10 * 1000;
- let _procTimer = null;
- let _procCpuTimes = new Map();
- let _procLastSample = 0;
- // The inference process's footprint at the last sample, in bytes --
- // what the between-batches restart check reads
- let _inferenceFootprint = 0;
-
- // One line of process usage for the debug log -- the main and inference
- // processes' footprint and CPU (in core-fractions, so several busy
- // threads read over 100%), and the memory still available -- so a
- // submitted debug log shows what indexing cost while it ran.
- async function _sampleProcesses() {
- let info = await ChromeUtils.requestProcInfo();
- let now = Date.now();
- let elapsedNS = _procLastSample ? (now - _procLastSample) * 1e6 : 0;
- _procLastSample = now;
- let inference = info.children.find(child => child.type == 'inference');
- _inferenceFootprint = inference ? inference.memory : 0;
- let procs = [
- { label: 'main', pid: info.pid, memory: info.memory, cpuTime: info.cpuTime }
- ];
- if (inference) {
- procs.push({
- label: 'inference',
- pid: inference.pid,
- memory: inference.memory,
- cpuTime: inference.cpuTime
- });
- }
- let parts = procs.map((proc) => {
- let cpu = '';
- let prev = _procCpuTimes.get(proc.pid);
- if (prev !== undefined && elapsedNS) {
- cpu = ` cpu ${Math.round((proc.cpuTime - prev) / elapsedNS * 100)}%`;
- }
- _procCpuTimes.set(proc.pid, proc.cpuTime);
- return `${proc.label} ${(proc.memory / 1024 / 1024).toFixed(0)} MB${cpu}`;
- });
- let available = _availableMemory();
- Zotero.debug('Embeddings: ' + parts.join(', ')
- + (available ? `, ${Math.round(available / 1024 / 1024)} MB available` : ''));
- }
-
- function _startProcMonitor() {
- if (_procTimer) {
- return;
- }
- _procCpuTimes = new Map();
- _procLastSample = 0;
- _inferenceFootprint = 0;
- _procTimer = setInterval(
- () => _sampleProcesses().catch(e => Zotero.logError(e)),
- PROC_SAMPLE_INTERVAL
- );
- }
-
- function _stopProcMonitor() {
- if (!_procTimer) {
- return;
- }
- clearInterval(_procTimer);
- _procTimer = null;
- }
+ // Time of the process sample the last cap restart acted on, so each
+ // sample triggers at most one
+ let _restartedOnSample = 0;
// Serialize model switches so rapid preference changes don't run their
// clear/prune/re-index steps concurrently.
@@ -1916,6 +1861,7 @@ Zotero.Embeddings.Indexing = new function () {
}
return;
}
+ Zotero.Embeddings.Diagnostics.recordMemoryPressure();
if (_tokenBudget <= DEGRADED_TOKEN_BUDGET_FLOOR) {
return;
}
@@ -1979,7 +1925,7 @@ Zotero.Embeddings.Indexing = new function () {
* @return {Boolean}
*/
function _hasMemoryToIndex() {
- let available = _availableMemory();
+ let available = Zotero.Embeddings.Diagnostics.getAvailableMemory();
if (available && available < MIN_AVAILABLE_MEMORY) {
Zotero.debug(`Embeddings: only ${Math.round(available / 1024 / 1024)} MB `
+ "available -- not indexing yet");
@@ -1988,20 +1934,6 @@ Zotero.Embeddings.Indexing = new function () {
return true;
}
- // Physical memory available right now, in bytes -- 0 when the platform
- // can't say
- function _availableMemory() {
- try {
- return Cc["@mozilla.org/ml-utils;1"]
- .getService(Ci.nsIMLUtils)
- .availablePhysicalMemory || 0;
- }
- catch (e) {
- Zotero.logError(e);
- return 0;
- }
- }
-
// Put an itemID on the queue its kind of work belongs to. An item the
// cache can't type goes to the regular queue, which routes it when it
// loads (see _drainItemQueue()).
@@ -2582,8 +2514,8 @@ Zotero.Embeddings.Indexing = new function () {
"INSERT INTO embeddings.itemEmbeddings "
+ "(itemID, chunkIndex, embedding, sourceHash, "
+ "startBlock, endBlock, startOffset, endOffset, "
- + "textCheck, sectionPart, sectionParts) "
- + "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
+ + "textCheck, sectionPart, sectionParts, tokens) "
+ + "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
[
entry.item.id,
chunkIndex,
@@ -2595,7 +2527,8 @@ Zotero.Embeddings.Indexing = new function () {
isAttachment ? chunk.endOffset ?? null : null,
isAttachment ? Zotero.Embeddings.textCheck(chunk.text) : null,
chunk.sectionPart ?? null,
- chunk.sectionParts ?? null
+ chunk.sectionParts ?? null,
+ chunk.tokens ?? null
],
{ debugParams: false }
);
@@ -2860,6 +2793,7 @@ Zotero.Embeddings.Indexing = new function () {
// halves fulltext indexing time versus item-level ordering.
units.sort((a, b) => a.entry.chunks[a.chunkIndex].tokens
- b.entry.chunks[b.chunkIndex].tokens);
+ Zotero.Embeddings.Diagnostics.startSlice(units.length);
let done = 0;
for (let i = 0; i < units.length;) {
@@ -2881,15 +2815,24 @@ Zotero.Embeddings.Indexing = new function () {
}
let batch = units.slice(i, i + count);
i += count;
+ let started = Date.now();
let vectors = await Zotero.Embeddings.embedPassages(
batch.map(unit => embedText(unit.entry.chunks[unit.chunkIndex]))
);
- // The arena's only reclaim is a restart (see INFERENCE_MEMORY_CAP);
- // zeroing the footprint holds the next check until a fresh sample
- if (_inferenceFootprint > INFERENCE_MEMORY_CAP) {
- let footprintMB = Math.round(_inferenceFootprint / 1024 / 1024);
- _inferenceFootprint = 0;
+ Zotero.Embeddings.Diagnostics.recordBatch({
+ chunks: batch.length,
+ tokens: batch.reduce((sum, unit) => sum + unit.entry.chunks[unit.chunkIndex].tokens, 0),
+ longest,
+ inferenceMs: Date.now() - started
+ });
+ // The arena's only reclaim is a restart (see INFERENCE_MEMORY_CAP)
+ let sample = Zotero.Embeddings.Diagnostics.getProcessSample();
+ if (sample?.inference?.memory > INFERENCE_MEMORY_CAP
+ && sample.time !== _restartedOnSample) {
+ _restartedOnSample = sample.time;
+ let footprintMB = Math.round(sample.inference.memory / 1024 / 1024);
await Zotero.Embeddings.shutdownEngine({ modelChanged: false });
+ Zotero.Embeddings.Diagnostics.recordRestart('memory');
Zotero.debug(`Embeddings: inference process at ${footprintMB} MB `
+ '-- engine restarted to release its memory');
}
@@ -2897,6 +2840,7 @@ Zotero.Embeddings.Indexing = new function () {
// between batches, never mid-request
else if (Zotero.Embeddings.engineThreadsStale()) {
await Zotero.Embeddings.shutdownEngine({ modelChanged: false });
+ Zotero.Embeddings.Diagnostics.recordRestart('threads');
Zotero.debug('Embeddings: engine restarted to apply new thread count');
}
let completed = [];
@@ -3001,6 +2945,20 @@ Zotero.Embeddings.Indexing = new function () {
extractionProgress: _extractionProgress,
items: _itemCounts,
chunks: _chunkCounts,
+ // Seconds until the fulltext work is embedded, at the current rate.
+ // Unknown while extraction is still adding to the total.
+ eta: _phase === 'extracting'
+ ? null
+ : Zotero.Embeddings.Diagnostics.estimateSeconds(_chunkCounts.total - _chunkCounts.done),
+ diagnostics: {
+ ...Zotero.Embeddings.Diagnostics.getStatus(),
+ tokenBudget: _tokenBudget,
+ engine: {
+ threads: this.getEngineThreads(),
+ optimalThreads: Zotero.ML.getOptimalConcurrency(),
+ boosts: [..._threadBoosts]
+ }
+ },
error: _lastError ? (_lastError.message || String(_lastError)) : null
};
};
@@ -3056,6 +3014,7 @@ Zotero.Embeddings.Indexing = new function () {
}
_itemCounts = { done: await _getIndexedItemCount(), total };
_chunkCounts = await _getChunkCounts();
+ await Zotero.Embeddings.Diagnostics.refreshChunkShape();
_lastCountRefresh = Date.now();
_emitProgress();
return Zotero.Embeddings.Indexing.getStatus();
@@ -3072,8 +3031,8 @@ Zotero.Embeddings.Indexing = new function () {
// Progress tick for a run's inner loops: refresh the chunk counts and
// emit, at most once per PROGRESS_EMIT_INTERVAL. The indexed item count
- // costs a scan of the index, so it's recomputed less often; the eligible
- // count comes from the last full refresh.
+ // and chunk shape each cost an index scan, so they're recomputed less
+ // often; the eligible count comes from the last full refresh.
async function _tick() {
let now = Date.now();
if (now - _lastTick < PROGRESS_EMIT_INTERVAL) {
@@ -3085,6 +3044,7 @@ Zotero.Embeddings.Indexing = new function () {
if (now - _lastCountRefresh >= COUNT_REFRESH_INTERVAL) {
_lastCountRefresh = now;
_itemCounts = { done: await _getIndexedItemCount(), total: _itemCounts.total };
+ await Zotero.Embeddings.Diagnostics.refreshChunkShape();
}
}
catch (e) {
@@ -3198,7 +3158,7 @@ Zotero.Embeddings.Indexing = new function () {
_stopping = false;
_lastError = null;
_lastTick = 0;
- _startProcMonitor();
+ Zotero.Embeddings.Diagnostics.startRun();
_startIdleWatch();
try {
await Zotero.Embeddings.initDB();
@@ -3247,11 +3207,11 @@ Zotero.Embeddings.Indexing = new function () {
}
finally {
_stopIdleWatch();
- _stopProcMonitor();
_indexing = false;
_phase = 'idle';
_downloadProgress = null;
_extractionProgress = null;
+ Zotero.Embeddings.Diagnostics.endRun();
// A run cut short by a stop or an error still reports what's
// stored
try {
@@ -3327,6 +3287,272 @@ Zotero.Embeddings.Indexing = new function () {
}
};
+/**
+ * Pipeline diagnostics, reported through Indexing.getStatus().
+ * Indexing records what happens -- engine batches, restarts, memory
+ * pressure, process samples, the slice in progress -- and this turns it
+ * into rates and shape summaries. Tokens are the chunker's estimates.
+ */
+Zotero.Embeddings.Diagnostics = new function () {
+ // Throughput window: at least a slice, since the in-slice length sort
+ // makes shorter windows swing
+ const RATE_WINDOW = 120 * 1000;
+ // A window this long is trusted for estimates
+ const ESTABLISHED_WINDOW = 30 * 1000;
+ // How often the main and inference processes are sampled during a run
+ const PROC_SAMPLE_INTERVAL = 10 * 1000;
+ let _samples = [];
+ let _run = _newRun();
+ let _slice = null;
+ let _chunkShape = null;
+ let _pressureEvents = 0;
+ let _procTimer = null;
+ let _procCpuTimes = new Map();
+ let _procLastSample = 0;
+ let _processSample = null;
+
+ function _newRun() {
+ return {
+ batches: 0,
+ chunks: 0,
+ tokens: 0,
+ padded: 0,
+ inferenceMs: 0,
+ restarts: { memory: 0, threads: 0 }
+ };
+ }
+
+ // Reset everything scoped to one indexing run and start sampling the
+ // processes
+ this.startRun = function () {
+ _samples = [];
+ _run = _newRun();
+ _pressureEvents = 0;
+ _slice = null;
+ if (!_procTimer) {
+ _procCpuTimes = new Map();
+ _procLastSample = 0;
+ _procTimer = setInterval(
+ () => _sampleProcesses().catch(e => Zotero.logError(e)),
+ PROC_SAMPLE_INTERVAL
+ );
+ }
+ };
+
+ this.endRun = function () {
+ _slice = null;
+ if (_procTimer) {
+ clearInterval(_procTimer);
+ _procTimer = null;
+ }
+ };
+
+ // Throughput over the last RATE_WINDOW, wall-clock -- so it includes
+ // commits and restarts -- or null before the first batch
+ function _getWindow() {
+ if (!_samples.length) {
+ return null;
+ }
+ let sum = key => _samples.reduce((total, sample) => total + sample[key], 0);
+ let span = Date.now() - _samples[0].time;
+ let seconds = Math.max(1, span / 1000);
+ let tokens = sum('tokens');
+ return {
+ chunksPerSecond: sum('chunks') / seconds,
+ tokensPerSecond: tokens / seconds,
+ paddingEfficiency: tokens / (sum('padded') || 1),
+ established: span >= ESTABLISHED_WINDOW
+ };
+ }
+
+ // Seconds to embed `remaining` chunks at the window's rate, or null when
+ // there's nothing left or the window isn't established
+ this.estimateSeconds = function (remaining) {
+ let window = _getWindow();
+ if (!window?.established || remaining <= 0) {
+ return null;
+ }
+ return remaining / window.chunksPerSecond;
+ };
+
+ // The last process sample: { time, available, main: { memory, cpu },
+ // inference: { memory, cpu } }, with memory in bytes and CPU in
+ // core-fractions (several busy threads read over 100%). Null before the
+ // first sample of a run; inference is absent when no engine is up.
+ this.getProcessSample = function () {
+ return _processSample;
+ };
+
+ // Physical memory available right now, in bytes -- 0 when the platform
+ // can't say
+ this.getAvailableMemory = function () {
+ try {
+ return Cc["@mozilla.org/ml-utils;1"]
+ .getService(Ci.nsIMLUtils)
+ .availablePhysicalMemory || 0;
+ }
+ catch (e) {
+ Zotero.logError(e);
+ return 0;
+ }
+ };
+
+
+ async function _sampleProcesses() {
+ let info = await ChromeUtils.requestProcInfo();
+ let now = Date.now();
+ let elapsedNS = _procLastSample ? (now - _procLastSample) * 1e6 : 0;
+ _procLastSample = now;
+ let inference = info.children.find(child => child.type == 'inference');
+ let procs = [
+ { label: 'main', pid: info.pid, memory: info.memory, cpuTime: info.cpuTime }
+ ];
+ if (inference) {
+ procs.push({
+ label: 'inference',
+ pid: inference.pid,
+ memory: inference.memory,
+ cpuTime: inference.cpuTime
+ });
+ }
+ let sample = { time: now, available: Zotero.Embeddings.Diagnostics.getAvailableMemory() };
+ for (let proc of procs) {
+ let prev = _procCpuTimes.get(proc.pid);
+ _procCpuTimes.set(proc.pid, proc.cpuTime);
+ sample[proc.label] = {
+ memory: proc.memory,
+ cpu: prev !== undefined && elapsedNS
+ ? Math.round((proc.cpuTime - prev) / elapsedNS * 100)
+ : null
+ };
+ }
+ _processSample = sample;
+ }
+
+ // A slice of `total` chunks is about to be embedded
+ this.startSlice = function (total) {
+ _slice = { done: 0, total };
+ };
+
+ // Record an engine batch that finished at `time` (now by default). Padded
+ // tokens are what the engine computed: every text as long as the longest.
+ this.recordBatch = function ({ chunks, tokens, longest, inferenceMs, time = Date.now() }) {
+ let sample = { time, chunks, tokens, padded: chunks * longest, inferenceMs };
+ _samples.push(sample);
+ while (_samples.length && _samples[0].time < sample.time - RATE_WINDOW) {
+ _samples.shift();
+ }
+ _run.batches++;
+ _run.chunks += chunks;
+ _run.tokens += tokens;
+ _run.padded += sample.padded;
+ _run.inferenceMs += inferenceMs;
+ if (_slice) {
+ _slice.done += chunks;
+ }
+ };
+
+ // @param {String} cause - 'memory' or 'threads'
+ this.recordRestart = function (cause) {
+ _run.restarts[cause]++;
+ };
+
+ this.recordMemoryPressure = function () {
+ _pressureEvents++;
+ };
+
+ // Recount the chunk shape from the database (see _getChunkShape())
+ this.refreshChunkShape = async function () {
+ _chunkShape = await _getChunkShape();
+ };
+
+ // The window is wall-clock throughput; the run's inference speed counts
+ // only time inside the engine, so the gap between them is overhead.
+ this.getStatus = function () {
+ let window = _getWindow();
+ let run = null;
+ if (_run.batches) {
+ let seconds = Math.max(0.001, _run.inferenceMs / 1000);
+ run = {
+ chunksPerSecond: _run.chunks / seconds,
+ tokensPerSecond: _run.tokens / seconds,
+ paddingEfficiency: _run.tokens / (_run.padded || 1),
+ batches: _run.batches,
+ chunksPerBatch: _run.chunks / _run.batches,
+ tokensPerBatch: _run.tokens / _run.batches
+ };
+ }
+ return {
+ window,
+ run,
+ restarts: _run.restarts,
+ pressureEvents: _pressureEvents,
+ processes: _processSample,
+ slice: _slice,
+ chunks: _chunkShape
+ };
+ };
+
+ // The shape of the stored attachment chunks -- sizes from the tokens
+ // index, chunks per document from the ledger -- for judging the
+ // chunker's output. Attachment rows are the ones with sectionParts;
+ // other item types' rows aren't the chunker's work.
+ async function _getChunkShape() {
+ let { BUDGET_TOKENS, MIN_TOKENS } = Zotero.Utilities.Internal.Chunking;
+ let sizeBounds = [MIN_TOKENS, BUDGET_TOKENS / 2, Math.round(BUDGET_TOKENS * 5 / 6), BUDGET_TOKENS];
+ let documentBounds = [1, 11, 51, 201];
+ let bucketSQL = (column, bounds) => bounds.map((bound, i) => (i
+ ? `SUM(${column} >= ${bounds[i - 1]} AND ${column} < ${bound}) AS b${i}`
+ : `SUM(${column} < ${bound}) AS b0`
+ )).concat(`SUM(${column} >= ${bounds[bounds.length - 1]}) AS b${bounds.length}`).join(', ');
+ let buckets = (row, bounds) => bounds.map((bound, i) => ({
+ from: i ? bounds[i - 1] : null,
+ to: bound,
+ count: row[`b${i}`] || 0
+ })).concat({ from: bounds[bounds.length - 1], to: null, count: row[`b${bounds.length}`] || 0 });
+ let median = async (sql, count) => (count
+ ? Zotero.DB.valueQueryAsync(sql + " LIMIT 1 OFFSET " + Math.floor(count / 2))
+ : 0);
+
+ let sizes = await Zotero.DB.rowQueryAsync(
+ "SELECT COUNT(*) AS count, COALESCE(SUM(tokens), 0) AS tokens, "
+ + bucketSQL('tokens', sizeBounds) + ", "
+ + "SUM(sectionParts > 1) AS split, "
+ + "SUM(sectionParts > 1 AND sectionPart = 1) AS splitSections, "
+ + "SUM(CASE WHEN sectionParts > 1 AND sectionPart = 1 THEN sectionParts ELSE 0 END) AS splitParts "
+ + "FROM embeddings.itemEmbeddings WHERE sectionParts IS NOT NULL"
+ );
+ let documents = await Zotero.DB.rowQueryAsync(
+ "SELECT COUNT(*) AS count, COALESCE(SUM(chunks), 0) AS chunks, "
+ + "COALESCE(MAX(chunks), 0) AS max, " + bucketSQL('chunks', documentBounds)
+ + " FROM embeddings.itemChunkCounts"
+ );
+ return {
+ sizes: {
+ count: sizes.count,
+ tokens: sizes.tokens,
+ mean: sizes.count ? sizes.tokens / sizes.count : 0,
+ median: await median(
+ "SELECT tokens FROM embeddings.itemEmbeddings WHERE sectionParts IS NOT NULL "
+ + "ORDER BY tokens",
+ sizes.count),
+ buckets: buckets(sizes, sizeBounds),
+ splitShare: sizes.count ? (sizes.split || 0) / sizes.count : 0,
+ partsPerSplitSection: sizes.splitSections ? sizes.splitParts / sizes.splitSections : 0
+ },
+ perDocument: {
+ count: documents.count,
+ mean: documents.count ? documents.chunks / documents.count : 0,
+ median: await median(
+ "SELECT chunks FROM embeddings.itemChunkCounts ORDER BY chunks", documents.count),
+ max: documents.max,
+ buckets: buckets(documents, documentBounds)
+ }
+ };
+ }
+};
+
+
/**
* Zotero.Embeddings.Calibration -- how a model's scoring numbers are derived.
*
diff --git a/scss/preferences/_advanced.scss b/scss/preferences/_advanced.scss
index b907aa8168..f5c7755075 100644
--- a/scss/preferences/_advanced.scss
+++ b/scss/preferences/_advanced.scss
@@ -119,6 +119,22 @@
}
}
+#semantic-search-diagnostics {
+ display: grid;
+ grid-template-columns: max-content 1fr;
+ column-gap: 12px;
+ row-gap: 2px;
+ margin-top: 10px;
+
+ label:nth-child(odd) {
+ color: var(--fill-secondary);
+ }
+
+ label:nth-child(even) {
+ white-space: normal;
+ }
+}
+
#db-maintenance-options {
display: flex;
gap: 6px;
diff --git a/test/tests/embeddingsTest.js b/test/tests/embeddingsTest.js
index e68c698089..8b10aa13d6 100644
--- a/test/tests/embeddingsTest.js
+++ b/test/tests/embeddingsTest.js
@@ -1051,6 +1051,28 @@ describe("Zotero.Embeddings", function () {
});
});
+ describe("Diagnostics", function () {
+ it("should estimate remaining time once the window spans long enough", function () {
+ let diagnostics = Zotero.Embeddings.Diagnostics;
+ try {
+ diagnostics.startRun();
+ let batch = { chunks: 10, tokens: 1000, longest: 100, inferenceMs: 100 };
+ diagnostics.recordBatch(batch);
+ // One batch spans no time, so the window isn't trusted yet
+ assert.isNull(diagnostics.estimateSeconds(300));
+ // A batch a minute ago makes the window a minute wide
+ diagnostics.startRun();
+ diagnostics.recordBatch({ ...batch, time: Date.now() - 60 * 1000 });
+ diagnostics.recordBatch(batch);
+ assert.approximately(diagnostics.estimateSeconds(300), 900, 1);
+ assert.isNull(diagnostics.estimateSeconds(0));
+ }
+ finally {
+ diagnostics.endRun();
+ }
+ });
+ });
+
describe("Calibration", function () {
// Stub the model rather than setting the pref: writing embeddings.model
// kicks off a real model switch, which clears the index and the stored
@@ -1772,6 +1794,73 @@ describe("Zotero.Embeddings", function () {
}
});
+ it("should report pipeline diagnostics", async function () {
+ this.timeout(60000);
+ let item = await createDataObject('item', { title: 'Parent of measured attachment' });
+ let attachment = await importPDFAttachment(item);
+ let vector = new Float32Array(4).fill(0.5);
+ let stubs = [
+ sinon.stub(Zotero.Embeddings, 'embedPassages')
+ .callsFake(async texts => texts.map(() => vector)),
+ sinon.stub(Zotero.Embeddings, 'isEnabled').returns(true),
+ sinon.stub(Zotero.Embeddings, 'getModelVersion').returns('test-model/1'),
+ sinon.stub(Zotero.Embeddings, 'isDownloaded').resolves(true),
+ sinon.stub(Zotero.Embeddings, 'download').resolves(),
+ sinon.stub(Zotero.Embeddings, 'ensureCalibration').resolves(),
+ sinon.stub(Zotero.Embeddings, 'getModelName').returns('bge-small-en-v1.5'),
+ sinon.stub(Zotero.SDT, 'ensure').resolves(true),
+ sinon.stub(Zotero.SDT, 'getSections').resolves({
+ ok: true,
+ sections: [
+ sdtSection('', 0, ['Owls hunt at night. '.repeat(200)]),
+ sdtSection('', 1, ['Hawks hunt by day. '.repeat(200)])
+ ]
+ })
+ ];
+ try {
+ Zotero.Prefs.set('embeddings.indexFulltext', true);
+ await Zotero.Embeddings.Indexing.startIndexing();
+ let { diagnostics } = await Zotero.Embeddings.Indexing.refreshStatus();
+
+ // Every stored chunk carries its size
+ assert.equal(await Zotero.DB.valueQueryAsync(
+ "SELECT COUNT(*) FROM embeddings.itemEmbeddings WHERE itemID=? AND tokens IS NULL",
+ attachment.id
+ ), 0);
+
+ // Rates come from the run's batches
+ assert.isAbove(diagnostics.run.batches, 0);
+ assert.isAbove(diagnostics.run.chunksPerSecond, 0);
+ assert.isAbove(diagnostics.run.tokensPerSecond, 0);
+ assert.isAbove(diagnostics.run.paddingEfficiency, 0);
+ assert.isAtMost(diagnostics.run.paddingEfficiency, 1);
+ assert.isAbove(diagnostics.window.chunksPerSecond, 0);
+ assert.isNull(diagnostics.slice);
+ assert.isAtLeast(diagnostics.engine.threads, 1);
+ // Nothing left to embed, so no estimate
+ assert.isNull((await Zotero.Embeddings.Indexing.refreshStatus()).eta);
+
+ // Chunk shape agrees with the tables
+ let { sizes, perDocument } = diagnostics.chunks;
+ assert.equal(sizes.count, await Zotero.DB.valueQueryAsync(
+ "SELECT COUNT(*) FROM embeddings.itemEmbeddings WHERE sectionParts IS NOT NULL"
+ ));
+ assert.equal(sizes.buckets.reduce((sum, b) => sum + b.count, 0), sizes.count);
+ assert.isAbove(sizes.median, 0);
+ assert.equal(perDocument.count, await Zotero.DB.valueQueryAsync(
+ "SELECT COUNT(*) FROM embeddings.itemChunkCounts"
+ ));
+ assert.equal(perDocument.buckets.reduce((sum, b) => sum + b.count, 0), perDocument.count);
+ assert.isAtLeast(perDocument.max, await Zotero.DB.valueQueryAsync(
+ "SELECT chunks FROM embeddings.itemChunkCounts WHERE itemID=?", attachment.id
+ ));
+ }
+ finally {
+ stubs.forEach(stub => stub.restore());
+ Zotero.Prefs.clear('embeddings.indexFulltext');
+ }
+ });
+
it("should index auxiliary chunks with words and drop bare labels", async function () {
this.timeout(60000);
let item = await createDataObject('item', { title: 'Parent of captioned attachment' });