display indexing metrics in the prefs pane

For easier evaluation of the progress and troubleshooting.

ETA: seconds to embed the remaining chunks at the current window rate
Throughput (2 min window): chunks/s and estimated tokens/s, wall-clock
Inference speed (run): chunks/s and tokens/s counting engine time only
Padding efficiency: real tokens over padded tokens, window and run
Batches (run): count, mean chunks and mean tokens per batch
Token budget and number of memory-pressure events this run
Engine threads: in use, runtime optimum, active boost reasons
Engine restarts (run): by cause, memory cap or thread change
Inference and main process: memory and CPU from the 10 s sample
Available system memory
Slice: chunks embedded out of the current slice
Chunks stored: count, total tokens, mean and median size
Chunk sizes: counts in five buckets bounded at 120, 384, 640 and 768 tokens
Split sections: share of chunks from split sections, mean parts per split
Chunks per document: document count, mean, median, max
Documents by chunk count: buckets at 0, 1–10, 11–50, 51–200, over 200
This commit is contained in:
Bogdan Abaev 2026-09-06 17:52:10 -07:00
parent c1d1f16fc0
commit 3789b95795
5 changed files with 488 additions and 93 deletions

View file

@ -231,6 +231,69 @@ Zotero_Preferences.Advanced = {
document.getElementById('semantic-search-attachments-row').hidden
= !Zotero.Prefs.get('embeddings.indexFulltext');
this._updateSemanticSearchBar('attachments', status.chunks);
this._updateSemanticSearchDiagnostics(status.diagnostics, status.eta);
},
// Key/value rows of pipeline diagnostics for developers, so the labels
// are plain English rather than localized
_updateSemanticSearchDiagnostics: function (diagnostics, eta) {
let n = (value, digits = 0) => (value ?? 0).toLocaleString(undefined, {
maximumFractionDigits: digits, minimumFractionDigits: digits
});
let pct = value => Math.round((value || 0) * 100) + '%';
let mb = bytes => n(bytes / 1024 / 1024) + ' MB';
let speed = rate => `${n(rate.chunksPerSecond, 1)} chunks/s, ${n(rate.tokensPerSecond)} tokens/s`;
let bucketList = (buckets, total, unit) => buckets.map(({ from, to, count }) => {
let range = from === null ? `< ${n(to)}` : (to === null ? `≥ ${n(from)}` : `${n(from)}–${n(to - 1)}`);
return `${range}${unit}: ${n(count)} (${pct(total ? count / total : 0)})`;
}).join(' · ');
let proc = p => (p ? `${mb(p.memory)}${p.cpu === null ? '' : `, CPU ${p.cpu}%`}` : '—');
let duration = (seconds) => {
let h = Math.floor(seconds / 3600);
let m = Math.floor(seconds % 3600 / 60);
return h ? `${h} h ${m} m` : `${m} m ${Math.floor(seconds % 60)} s`;
};
let rows = [];
let { window, run, engine, processes, slice, chunks } = diagnostics;
rows.push(['ETA', eta === null ? '—' : duration(eta)]);
rows.push(['Throughput (2 min)', window ? speed(window) : '—']);
rows.push(['Inference speed (run)', run ? speed(run) : '—']);
rows.push(['Padding efficiency', window || run
? `${window ? pct(window.paddingEfficiency) : '—'} (2 min), ${run ? pct(run.paddingEfficiency) : '—'} (run)`
: '—']);
rows.push(['Batches (run)', run
? `${n(run.batches)} · ${n(run.chunksPerBatch, 1)} chunks · ${n(run.tokensPerBatch)} tokens avg`
: '—']);
rows.push(['Token budget', `${n(diagnostics.tokenBudget)} · ${n(diagnostics.pressureEvents)} memory-pressure events`]);
rows.push(['Engine threads', `${engine.threads} of ${engine.optimalThreads}`
+ (engine.boosts.length ? ` (boost: ${engine.boosts.join(', ')})` : '')]);
rows.push(['Engine restarts (run)', `memory ${diagnostics.restarts.memory}, threads ${diagnostics.restarts.threads}`]);
rows.push(['Inference process', proc(processes?.inference)]);
rows.push(['Main process', proc(processes?.main)]);
rows.push(['Available memory', processes?.available ? mb(processes.available) : '—']);
rows.push(['Slice', slice ? `${n(slice.done)} / ${n(slice.total)} chunks` : '—']);
if (chunks) {
let { sizes, perDocument } = chunks;
rows.push(['Chunks stored', `${n(sizes.count)} · ${n(sizes.tokens)} tokens · mean ${n(sizes.mean)} · median ${n(sizes.median)}`]);
rows.push(['Chunk sizes', bucketList(sizes.buckets, sizes.count, '')]);
rows.push(['Split sections', `${pct(sizes.splitShare)} of chunks · ${n(sizes.partsPerSplitSection, 1)} parts per split section`]);
rows.push(['Chunks per document', `${n(perDocument.count)} documents · mean ${n(perDocument.mean, 1)} · median ${n(perDocument.median)} · max ${n(perDocument.max)}`]);
rows.push(['Documents by chunks', bucketList(perDocument.buckets, perDocument.count, '')]);
}
let box = document.getElementById('semantic-search-diagnostics');
while (box.childElementCount < rows.length * 2) {
box.append(document.createXULElement('label'), document.createXULElement('label'));
}
while (box.childElementCount > rows.length * 2) {
box.lastElementChild.remove();
}
rows.forEach(([label, value], i) => {
box.children[i * 2].value = label;
box.children[i * 2 + 1].value = value;
});
},

View file

@ -355,6 +355,7 @@
<label id="semantic-search-attachments-value"/>
</html:div>
</html:div>
<html:div id="semantic-search-diagnostics"/>
</vbox>
</groupbox>
</vbox>

View file

@ -287,7 +287,7 @@ Zotero.Embeddings = new function () {
// Schema version of the attached embeddings database. The tables are only
// created when this is bumped (_setUpDB() drops and recreates everything),
// so any schema change needs a bump.
const _dbVersion = 4;
const _dbVersion = 5;
let _dbInitPromise = null;
let _dbHooksRegistered = false;
@ -391,9 +391,16 @@ Zotero.Embeddings = new function () {
+ " textCheck TEXT,\n"
+ " sectionPart INTEGER,\n"
+ " sectionParts INTEGER,\n"
+ " tokens INTEGER,\n"
+ " PRIMARY KEY (itemID, chunkIndex)\n"
+ ")"
);
// Chunk-shape diagnostics read this index alone, never the rows with
// their vectors (see Indexing._getChunkShape())
await Zotero.DB.queryAsync(
"CREATE INDEX embeddings.itemEmbeddings_tokens "
+ "ON itemEmbeddings (tokens, sectionParts, sectionPart)"
);
// The localUserKey the vectors were built against and the model that
// produced them
await Zotero.DB.queryAsync(
@ -1688,71 +1695,9 @@ Zotero.Embeddings.Indexing = new function () {
// gets. Restarting the engine is the only reclaim and costs about a
// second, so past this footprint it's restarted between batches.
const INFERENCE_MEMORY_CAP = 1.5 * 1024 * 1024 * 1024;
// How often process usage is sampled and logged during a run
const PROC_SAMPLE_INTERVAL = 10 * 1000;
let _procTimer = null;
let _procCpuTimes = new Map();
let _procLastSample = 0;
// The inference process's footprint at the last sample, in bytes --
// what the between-batches restart check reads
let _inferenceFootprint = 0;
// One line of process usage for the debug log -- the main and inference
// processes' footprint and CPU (in core-fractions, so several busy
// threads read over 100%), and the memory still available -- so a
// submitted debug log shows what indexing cost while it ran.
async function _sampleProcesses() {
let info = await ChromeUtils.requestProcInfo();
let now = Date.now();
let elapsedNS = _procLastSample ? (now - _procLastSample) * 1e6 : 0;
_procLastSample = now;
let inference = info.children.find(child => child.type == 'inference');
_inferenceFootprint = inference ? inference.memory : 0;
let procs = [
{ label: 'main', pid: info.pid, memory: info.memory, cpuTime: info.cpuTime }
];
if (inference) {
procs.push({
label: 'inference',
pid: inference.pid,
memory: inference.memory,
cpuTime: inference.cpuTime
});
}
let parts = procs.map((proc) => {
let cpu = '';
let prev = _procCpuTimes.get(proc.pid);
if (prev !== undefined && elapsedNS) {
cpu = ` cpu ${Math.round((proc.cpuTime - prev) / elapsedNS * 100)}%`;
}
_procCpuTimes.set(proc.pid, proc.cpuTime);
return `${proc.label} ${(proc.memory / 1024 / 1024).toFixed(0)} MB${cpu}`;
});
let available = _availableMemory();
Zotero.debug('Embeddings: ' + parts.join(', ')
+ (available ? `, ${Math.round(available / 1024 / 1024)} MB available` : ''));
}
function _startProcMonitor() {
if (_procTimer) {
return;
}
_procCpuTimes = new Map();
_procLastSample = 0;
_inferenceFootprint = 0;
_procTimer = setInterval(
() => _sampleProcesses().catch(e => Zotero.logError(e)),
PROC_SAMPLE_INTERVAL
);
}
function _stopProcMonitor() {
if (!_procTimer) {
return;
}
clearInterval(_procTimer);
_procTimer = null;
}
// Time of the process sample the last cap restart acted on, so each
// sample triggers at most one
let _restartedOnSample = 0;
// Serialize model switches so rapid preference changes don't run their
// clear/prune/re-index steps concurrently.
@ -1916,6 +1861,7 @@ Zotero.Embeddings.Indexing = new function () {
}
return;
}
Zotero.Embeddings.Diagnostics.recordMemoryPressure();
if (_tokenBudget <= DEGRADED_TOKEN_BUDGET_FLOOR) {
return;
}
@ -1979,7 +1925,7 @@ Zotero.Embeddings.Indexing = new function () {
* @return {Boolean}
*/
function _hasMemoryToIndex() {
let available = _availableMemory();
let available = Zotero.Embeddings.Diagnostics.getAvailableMemory();
if (available && available < MIN_AVAILABLE_MEMORY) {
Zotero.debug(`Embeddings: only ${Math.round(available / 1024 / 1024)} MB `
+ "available -- not indexing yet");
@ -1988,20 +1934,6 @@ Zotero.Embeddings.Indexing = new function () {
return true;
}
// Physical memory available right now, in bytes -- 0 when the platform
// can't say
function _availableMemory() {
try {
return Cc["@mozilla.org/ml-utils;1"]
.getService(Ci.nsIMLUtils)
.availablePhysicalMemory || 0;
}
catch (e) {
Zotero.logError(e);
return 0;
}
}
// Put an itemID on the queue its kind of work belongs to. An item the
// cache can't type goes to the regular queue, which routes it when it
// loads (see _drainItemQueue()).
@ -2582,8 +2514,8 @@ Zotero.Embeddings.Indexing = new function () {
"INSERT INTO embeddings.itemEmbeddings "
+ "(itemID, chunkIndex, embedding, sourceHash, "
+ "startBlock, endBlock, startOffset, endOffset, "
+ "textCheck, sectionPart, sectionParts) "
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
+ "textCheck, sectionPart, sectionParts, tokens) "
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
[
entry.item.id,
chunkIndex,
@ -2595,7 +2527,8 @@ Zotero.Embeddings.Indexing = new function () {
isAttachment ? chunk.endOffset ?? null : null,
isAttachment ? Zotero.Embeddings.textCheck(chunk.text) : null,
chunk.sectionPart ?? null,
chunk.sectionParts ?? null
chunk.sectionParts ?? null,
chunk.tokens ?? null
],
{ debugParams: false }
);
@ -2860,6 +2793,7 @@ Zotero.Embeddings.Indexing = new function () {
// halves fulltext indexing time versus item-level ordering.
units.sort((a, b) => a.entry.chunks[a.chunkIndex].tokens
- b.entry.chunks[b.chunkIndex].tokens);
Zotero.Embeddings.Diagnostics.startSlice(units.length);
let done = 0;
for (let i = 0; i < units.length;) {
@ -2881,15 +2815,24 @@ Zotero.Embeddings.Indexing = new function () {
}
let batch = units.slice(i, i + count);
i += count;
let started = Date.now();
let vectors = await Zotero.Embeddings.embedPassages(
batch.map(unit => embedText(unit.entry.chunks[unit.chunkIndex]))
);
// The arena's only reclaim is a restart (see INFERENCE_MEMORY_CAP);
// zeroing the footprint holds the next check until a fresh sample
if (_inferenceFootprint > INFERENCE_MEMORY_CAP) {
let footprintMB = Math.round(_inferenceFootprint / 1024 / 1024);
_inferenceFootprint = 0;
Zotero.Embeddings.Diagnostics.recordBatch({
chunks: batch.length,
tokens: batch.reduce((sum, unit) => sum + unit.entry.chunks[unit.chunkIndex].tokens, 0),
longest,
inferenceMs: Date.now() - started
});
// The arena's only reclaim is a restart (see INFERENCE_MEMORY_CAP)
let sample = Zotero.Embeddings.Diagnostics.getProcessSample();
if (sample?.inference?.memory > INFERENCE_MEMORY_CAP
&& sample.time !== _restartedOnSample) {
_restartedOnSample = sample.time;
let footprintMB = Math.round(sample.inference.memory / 1024 / 1024);
await Zotero.Embeddings.shutdownEngine({ modelChanged: false });
Zotero.Embeddings.Diagnostics.recordRestart('memory');
Zotero.debug(`Embeddings: inference process at ${footprintMB} MB `
+ '-- engine restarted to release its memory');
}
@ -2897,6 +2840,7 @@ Zotero.Embeddings.Indexing = new function () {
// between batches, never mid-request
else if (Zotero.Embeddings.engineThreadsStale()) {
await Zotero.Embeddings.shutdownEngine({ modelChanged: false });
Zotero.Embeddings.Diagnostics.recordRestart('threads');
Zotero.debug('Embeddings: engine restarted to apply new thread count');
}
let completed = [];
@ -3001,6 +2945,20 @@ Zotero.Embeddings.Indexing = new function () {
extractionProgress: _extractionProgress,
items: _itemCounts,
chunks: _chunkCounts,
// Seconds until the fulltext work is embedded, at the current rate.
// Unknown while extraction is still adding to the total.
eta: _phase === 'extracting'
? null
: Zotero.Embeddings.Diagnostics.estimateSeconds(_chunkCounts.total - _chunkCounts.done),
diagnostics: {
...Zotero.Embeddings.Diagnostics.getStatus(),
tokenBudget: _tokenBudget,
engine: {
threads: this.getEngineThreads(),
optimalThreads: Zotero.ML.getOptimalConcurrency(),
boosts: [..._threadBoosts]
}
},
error: _lastError ? (_lastError.message || String(_lastError)) : null
};
};
@ -3056,6 +3014,7 @@ Zotero.Embeddings.Indexing = new function () {
}
_itemCounts = { done: await _getIndexedItemCount(), total };
_chunkCounts = await _getChunkCounts();
await Zotero.Embeddings.Diagnostics.refreshChunkShape();
_lastCountRefresh = Date.now();
_emitProgress();
return Zotero.Embeddings.Indexing.getStatus();
@ -3072,8 +3031,8 @@ Zotero.Embeddings.Indexing = new function () {
// Progress tick for a run's inner loops: refresh the chunk counts and
// emit, at most once per PROGRESS_EMIT_INTERVAL. The indexed item count
// costs a scan of the index, so it's recomputed less often; the eligible
// count comes from the last full refresh.
// and chunk shape each cost an index scan, so they're recomputed less
// often; the eligible count comes from the last full refresh.
async function _tick() {
let now = Date.now();
if (now - _lastTick < PROGRESS_EMIT_INTERVAL) {
@ -3085,6 +3044,7 @@ Zotero.Embeddings.Indexing = new function () {
if (now - _lastCountRefresh >= COUNT_REFRESH_INTERVAL) {
_lastCountRefresh = now;
_itemCounts = { done: await _getIndexedItemCount(), total: _itemCounts.total };
await Zotero.Embeddings.Diagnostics.refreshChunkShape();
}
}
catch (e) {
@ -3198,7 +3158,7 @@ Zotero.Embeddings.Indexing = new function () {
_stopping = false;
_lastError = null;
_lastTick = 0;
_startProcMonitor();
Zotero.Embeddings.Diagnostics.startRun();
_startIdleWatch();
try {
await Zotero.Embeddings.initDB();
@ -3247,11 +3207,11 @@ Zotero.Embeddings.Indexing = new function () {
}
finally {
_stopIdleWatch();
_stopProcMonitor();
_indexing = false;
_phase = 'idle';
_downloadProgress = null;
_extractionProgress = null;
Zotero.Embeddings.Diagnostics.endRun();
// A run cut short by a stop or an error still reports what's
// stored
try {
@ -3327,6 +3287,272 @@ Zotero.Embeddings.Indexing = new function () {
}
};
/**
* Pipeline diagnostics, reported through Indexing.getStatus().
* Indexing records what happens -- engine batches, restarts, memory
* pressure, process samples, the slice in progress -- and this turns it
* into rates and shape summaries. Tokens are the chunker's estimates.
*/
Zotero.Embeddings.Diagnostics = new function () {
// Throughput window: at least a slice, since the in-slice length sort
// makes shorter windows swing
const RATE_WINDOW = 120 * 1000;
// A window this long is trusted for estimates
const ESTABLISHED_WINDOW = 30 * 1000;
// How often the main and inference processes are sampled during a run
const PROC_SAMPLE_INTERVAL = 10 * 1000;
let _samples = [];
let _run = _newRun();
let _slice = null;
let _chunkShape = null;
let _pressureEvents = 0;
let _procTimer = null;
let _procCpuTimes = new Map();
let _procLastSample = 0;
let _processSample = null;
function _newRun() {
return {
batches: 0,
chunks: 0,
tokens: 0,
padded: 0,
inferenceMs: 0,
restarts: { memory: 0, threads: 0 }
};
}
// Reset everything scoped to one indexing run and start sampling the
// processes
this.startRun = function () {
_samples = [];
_run = _newRun();
_pressureEvents = 0;
_slice = null;
if (!_procTimer) {
_procCpuTimes = new Map();
_procLastSample = 0;
_procTimer = setInterval(
() => _sampleProcesses().catch(e => Zotero.logError(e)),
PROC_SAMPLE_INTERVAL
);
}
};
this.endRun = function () {
_slice = null;
if (_procTimer) {
clearInterval(_procTimer);
_procTimer = null;
}
};
// Throughput over the last RATE_WINDOW, wall-clock -- so it includes
// commits and restarts -- or null before the first batch
function _getWindow() {
if (!_samples.length) {
return null;
}
let sum = key => _samples.reduce((total, sample) => total + sample[key], 0);
let span = Date.now() - _samples[0].time;
let seconds = Math.max(1, span / 1000);
let tokens = sum('tokens');
return {
chunksPerSecond: sum('chunks') / seconds,
tokensPerSecond: tokens / seconds,
paddingEfficiency: tokens / (sum('padded') || 1),
established: span >= ESTABLISHED_WINDOW
};
}
// Seconds to embed `remaining` chunks at the window's rate, or null when
// there's nothing left or the window isn't established
this.estimateSeconds = function (remaining) {
let window = _getWindow();
if (!window?.established || remaining <= 0) {
return null;
}
return remaining / window.chunksPerSecond;
};
// The last process sample: { time, available, main: { memory, cpu },
// inference: { memory, cpu } }, with memory in bytes and CPU in
// core-fractions (several busy threads read over 100%). Null before the
// first sample of a run; inference is absent when no engine is up.
this.getProcessSample = function () {
return _processSample;
};
// Physical memory available right now, in bytes -- 0 when the platform
// can't say
this.getAvailableMemory = function () {
try {
return Cc["@mozilla.org/ml-utils;1"]
.getService(Ci.nsIMLUtils)
.availablePhysicalMemory || 0;
}
catch (e) {
Zotero.logError(e);
return 0;
}
};
async function _sampleProcesses() {
let info = await ChromeUtils.requestProcInfo();
let now = Date.now();
let elapsedNS = _procLastSample ? (now - _procLastSample) * 1e6 : 0;
_procLastSample = now;
let inference = info.children.find(child => child.type == 'inference');
let procs = [
{ label: 'main', pid: info.pid, memory: info.memory, cpuTime: info.cpuTime }
];
if (inference) {
procs.push({
label: 'inference',
pid: inference.pid,
memory: inference.memory,
cpuTime: inference.cpuTime
});
}
let sample = { time: now, available: Zotero.Embeddings.Diagnostics.getAvailableMemory() };
for (let proc of procs) {
let prev = _procCpuTimes.get(proc.pid);
_procCpuTimes.set(proc.pid, proc.cpuTime);
sample[proc.label] = {
memory: proc.memory,
cpu: prev !== undefined && elapsedNS
? Math.round((proc.cpuTime - prev) / elapsedNS * 100)
: null
};
}
_processSample = sample;
}
// A slice of `total` chunks is about to be embedded
this.startSlice = function (total) {
_slice = { done: 0, total };
};
// Record an engine batch that finished at `time` (now by default). Padded
// tokens are what the engine computed: every text as long as the longest.
this.recordBatch = function ({ chunks, tokens, longest, inferenceMs, time = Date.now() }) {
let sample = { time, chunks, tokens, padded: chunks * longest, inferenceMs };
_samples.push(sample);
while (_samples.length && _samples[0].time < sample.time - RATE_WINDOW) {
_samples.shift();
}
_run.batches++;
_run.chunks += chunks;
_run.tokens += tokens;
_run.padded += sample.padded;
_run.inferenceMs += inferenceMs;
if (_slice) {
_slice.done += chunks;
}
};
// @param {String} cause - 'memory' or 'threads'
this.recordRestart = function (cause) {
_run.restarts[cause]++;
};
this.recordMemoryPressure = function () {
_pressureEvents++;
};
// Recount the chunk shape from the database (see _getChunkShape())
this.refreshChunkShape = async function () {
_chunkShape = await _getChunkShape();
};
// The window is wall-clock throughput; the run's inference speed counts
// only time inside the engine, so the gap between them is overhead.
this.getStatus = function () {
let window = _getWindow();
let run = null;
if (_run.batches) {
let seconds = Math.max(0.001, _run.inferenceMs / 1000);
run = {
chunksPerSecond: _run.chunks / seconds,
tokensPerSecond: _run.tokens / seconds,
paddingEfficiency: _run.tokens / (_run.padded || 1),
batches: _run.batches,
chunksPerBatch: _run.chunks / _run.batches,
tokensPerBatch: _run.tokens / _run.batches
};
}
return {
window,
run,
restarts: _run.restarts,
pressureEvents: _pressureEvents,
processes: _processSample,
slice: _slice,
chunks: _chunkShape
};
};
// The shape of the stored attachment chunks -- sizes from the tokens
// index, chunks per document from the ledger -- for judging the
// chunker's output. Attachment rows are the ones with sectionParts;
// other item types' rows aren't the chunker's work.
async function _getChunkShape() {
let { BUDGET_TOKENS, MIN_TOKENS } = Zotero.Utilities.Internal.Chunking;
let sizeBounds = [MIN_TOKENS, BUDGET_TOKENS / 2, Math.round(BUDGET_TOKENS * 5 / 6), BUDGET_TOKENS];
let documentBounds = [1, 11, 51, 201];
let bucketSQL = (column, bounds) => bounds.map((bound, i) => (i
? `SUM(${column} >= ${bounds[i - 1]} AND ${column} < ${bound}) AS b${i}`
: `SUM(${column} < ${bound}) AS b0`
)).concat(`SUM(${column} >= ${bounds[bounds.length - 1]}) AS b${bounds.length}`).join(', ');
let buckets = (row, bounds) => bounds.map((bound, i) => ({
from: i ? bounds[i - 1] : null,
to: bound,
count: row[`b${i}`] || 0
})).concat({ from: bounds[bounds.length - 1], to: null, count: row[`b${bounds.length}`] || 0 });
let median = async (sql, count) => (count
? Zotero.DB.valueQueryAsync(sql + " LIMIT 1 OFFSET " + Math.floor(count / 2))
: 0);
let sizes = await Zotero.DB.rowQueryAsync(
"SELECT COUNT(*) AS count, COALESCE(SUM(tokens), 0) AS tokens, "
+ bucketSQL('tokens', sizeBounds) + ", "
+ "SUM(sectionParts > 1) AS split, "
+ "SUM(sectionParts > 1 AND sectionPart = 1) AS splitSections, "
+ "SUM(CASE WHEN sectionParts > 1 AND sectionPart = 1 THEN sectionParts ELSE 0 END) AS splitParts "
+ "FROM embeddings.itemEmbeddings WHERE sectionParts IS NOT NULL"
);
let documents = await Zotero.DB.rowQueryAsync(
"SELECT COUNT(*) AS count, COALESCE(SUM(chunks), 0) AS chunks, "
+ "COALESCE(MAX(chunks), 0) AS max, " + bucketSQL('chunks', documentBounds)
+ " FROM embeddings.itemChunkCounts"
);
return {
sizes: {
count: sizes.count,
tokens: sizes.tokens,
mean: sizes.count ? sizes.tokens / sizes.count : 0,
median: await median(
"SELECT tokens FROM embeddings.itemEmbeddings WHERE sectionParts IS NOT NULL "
+ "ORDER BY tokens",
sizes.count),
buckets: buckets(sizes, sizeBounds),
splitShare: sizes.count ? (sizes.split || 0) / sizes.count : 0,
partsPerSplitSection: sizes.splitSections ? sizes.splitParts / sizes.splitSections : 0
},
perDocument: {
count: documents.count,
mean: documents.count ? documents.chunks / documents.count : 0,
median: await median(
"SELECT chunks FROM embeddings.itemChunkCounts ORDER BY chunks", documents.count),
max: documents.max,
buckets: buckets(documents, documentBounds)
}
};
}
};
/**
* Zotero.Embeddings.Calibration -- how a model's scoring numbers are derived.
*

View file

@ -119,6 +119,22 @@
}
}
#semantic-search-diagnostics {
display: grid;
grid-template-columns: max-content 1fr;
column-gap: 12px;
row-gap: 2px;
margin-top: 10px;
label:nth-child(odd) {
color: var(--fill-secondary);
}
label:nth-child(even) {
white-space: normal;
}
}
#db-maintenance-options {
display: flex;
gap: 6px;

View file

@ -1051,6 +1051,28 @@ describe("Zotero.Embeddings", function () {
});
});
describe("Diagnostics", function () {
it("should estimate remaining time once the window spans long enough", function () {
let diagnostics = Zotero.Embeddings.Diagnostics;
try {
diagnostics.startRun();
let batch = { chunks: 10, tokens: 1000, longest: 100, inferenceMs: 100 };
diagnostics.recordBatch(batch);
// One batch spans no time, so the window isn't trusted yet
assert.isNull(diagnostics.estimateSeconds(300));
// A batch a minute ago makes the window a minute wide
diagnostics.startRun();
diagnostics.recordBatch({ ...batch, time: Date.now() - 60 * 1000 });
diagnostics.recordBatch(batch);
assert.approximately(diagnostics.estimateSeconds(300), 900, 1);
assert.isNull(diagnostics.estimateSeconds(0));
}
finally {
diagnostics.endRun();
}
});
});
describe("Calibration", function () {
// Stub the model rather than setting the pref: writing embeddings.model
// kicks off a real model switch, which clears the index and the stored
@ -1772,6 +1794,73 @@ describe("Zotero.Embeddings", function () {
}
});
it("should report pipeline diagnostics", async function () {
this.timeout(60000);
let item = await createDataObject('item', { title: 'Parent of measured attachment' });
let attachment = await importPDFAttachment(item);
let vector = new Float32Array(4).fill(0.5);
let stubs = [
sinon.stub(Zotero.Embeddings, 'embedPassages')
.callsFake(async texts => texts.map(() => vector)),
sinon.stub(Zotero.Embeddings, 'isEnabled').returns(true),
sinon.stub(Zotero.Embeddings, 'getModelVersion').returns('test-model/1'),
sinon.stub(Zotero.Embeddings, 'isDownloaded').resolves(true),
sinon.stub(Zotero.Embeddings, 'download').resolves(),
sinon.stub(Zotero.Embeddings, 'ensureCalibration').resolves(),
sinon.stub(Zotero.Embeddings, 'getModelName').returns('bge-small-en-v1.5'),
sinon.stub(Zotero.SDT, 'ensure').resolves(true),
sinon.stub(Zotero.SDT, 'getSections').resolves({
ok: true,
sections: [
sdtSection('', 0, ['Owls hunt at night. '.repeat(200)]),
sdtSection('', 1, ['Hawks hunt by day. '.repeat(200)])
]
})
];
try {
Zotero.Prefs.set('embeddings.indexFulltext', true);
await Zotero.Embeddings.Indexing.startIndexing();
let { diagnostics } = await Zotero.Embeddings.Indexing.refreshStatus();
// Every stored chunk carries its size
assert.equal(await Zotero.DB.valueQueryAsync(
"SELECT COUNT(*) FROM embeddings.itemEmbeddings WHERE itemID=? AND tokens IS NULL",
attachment.id
), 0);
// Rates come from the run's batches
assert.isAbove(diagnostics.run.batches, 0);
assert.isAbove(diagnostics.run.chunksPerSecond, 0);
assert.isAbove(diagnostics.run.tokensPerSecond, 0);
assert.isAbove(diagnostics.run.paddingEfficiency, 0);
assert.isAtMost(diagnostics.run.paddingEfficiency, 1);
assert.isAbove(diagnostics.window.chunksPerSecond, 0);
assert.isNull(diagnostics.slice);
assert.isAtLeast(diagnostics.engine.threads, 1);
// Nothing left to embed, so no estimate
assert.isNull((await Zotero.Embeddings.Indexing.refreshStatus()).eta);
// Chunk shape agrees with the tables
let { sizes, perDocument } = diagnostics.chunks;
assert.equal(sizes.count, await Zotero.DB.valueQueryAsync(
"SELECT COUNT(*) FROM embeddings.itemEmbeddings WHERE sectionParts IS NOT NULL"
));
assert.equal(sizes.buckets.reduce((sum, b) => sum + b.count, 0), sizes.count);
assert.isAbove(sizes.median, 0);
assert.equal(perDocument.count, await Zotero.DB.valueQueryAsync(
"SELECT COUNT(*) FROM embeddings.itemChunkCounts"
));
assert.equal(perDocument.buckets.reduce((sum, b) => sum + b.count, 0), perDocument.count);
assert.isAtLeast(perDocument.max, await Zotero.DB.valueQueryAsync(
"SELECT chunks FROM embeddings.itemChunkCounts WHERE itemID=?", attachment.id
));
}
finally {
stubs.forEach(stub => stub.restore());
Zotero.Prefs.clear('embeddings.indexFulltext');
}
});
it("should index auxiliary chunks with words and drop bare labels", async function () {
this.timeout(60000);
let item = await createDataObject('item', { title: 'Parent of captioned attachment' });