mirror of
https://github.com/zotero/zotero.git
synced 2026-10-08 03:08:19 +00:00
display indexing metrics in the prefs pane
For easier evaluation of the progress and troubleshooting. ETA: seconds to embed the remaining chunks at the current window rate Throughput (2 min window): chunks/s and estimated tokens/s, wall-clock Inference speed (run): chunks/s and tokens/s counting engine time only Padding efficiency: real tokens over padded tokens, window and run Batches (run): count, mean chunks and mean tokens per batch Token budget and number of memory-pressure events this run Engine threads: in use, runtime optimum, active boost reasons Engine restarts (run): by cause, memory cap or thread change Inference and main process: memory and CPU from the 10 s sample Available system memory Slice: chunks embedded out of the current slice Chunks stored: count, total tokens, mean and median size Chunk sizes: counts in five buckets bounded at 120, 384, 640 and 768 tokens Split sections: share of chunks from split sections, mean parts per split Chunks per document: document count, mean, median, max Documents by chunk count: buckets at 0, 1–10, 11–50, 51–200, over 200
This commit is contained in:
parent
c1d1f16fc0
commit
3789b95795
5 changed files with 488 additions and 93 deletions
|
|
@ -231,6 +231,69 @@ Zotero_Preferences.Advanced = {
|
|||
document.getElementById('semantic-search-attachments-row').hidden
|
||||
= !Zotero.Prefs.get('embeddings.indexFulltext');
|
||||
this._updateSemanticSearchBar('attachments', status.chunks);
|
||||
this._updateSemanticSearchDiagnostics(status.diagnostics, status.eta);
|
||||
},
|
||||
|
||||
|
||||
// Key/value rows of pipeline diagnostics for developers, so the labels
|
||||
// are plain English rather than localized
|
||||
_updateSemanticSearchDiagnostics: function (diagnostics, eta) {
|
||||
let n = (value, digits = 0) => (value ?? 0).toLocaleString(undefined, {
|
||||
maximumFractionDigits: digits, minimumFractionDigits: digits
|
||||
});
|
||||
let pct = value => Math.round((value || 0) * 100) + '%';
|
||||
let mb = bytes => n(bytes / 1024 / 1024) + ' MB';
|
||||
let speed = rate => `${n(rate.chunksPerSecond, 1)} chunks/s, ${n(rate.tokensPerSecond)} tokens/s`;
|
||||
let bucketList = (buckets, total, unit) => buckets.map(({ from, to, count }) => {
|
||||
let range = from === null ? `< ${n(to)}` : (to === null ? `≥ ${n(from)}` : `${n(from)}–${n(to - 1)}`);
|
||||
return `${range}${unit}: ${n(count)} (${pct(total ? count / total : 0)})`;
|
||||
}).join(' · ');
|
||||
let proc = p => (p ? `${mb(p.memory)}${p.cpu === null ? '' : `, CPU ${p.cpu}%`}` : '—');
|
||||
let duration = (seconds) => {
|
||||
let h = Math.floor(seconds / 3600);
|
||||
let m = Math.floor(seconds % 3600 / 60);
|
||||
return h ? `${h} h ${m} m` : `${m} m ${Math.floor(seconds % 60)} s`;
|
||||
};
|
||||
|
||||
let rows = [];
|
||||
let { window, run, engine, processes, slice, chunks } = diagnostics;
|
||||
rows.push(['ETA', eta === null ? '—' : duration(eta)]);
|
||||
rows.push(['Throughput (2 min)', window ? speed(window) : '—']);
|
||||
rows.push(['Inference speed (run)', run ? speed(run) : '—']);
|
||||
rows.push(['Padding efficiency', window || run
|
||||
? `${window ? pct(window.paddingEfficiency) : '—'} (2 min), ${run ? pct(run.paddingEfficiency) : '—'} (run)`
|
||||
: '—']);
|
||||
rows.push(['Batches (run)', run
|
||||
? `${n(run.batches)} · ${n(run.chunksPerBatch, 1)} chunks · ${n(run.tokensPerBatch)} tokens avg`
|
||||
: '—']);
|
||||
rows.push(['Token budget', `${n(diagnostics.tokenBudget)} · ${n(diagnostics.pressureEvents)} memory-pressure events`]);
|
||||
rows.push(['Engine threads', `${engine.threads} of ${engine.optimalThreads}`
|
||||
+ (engine.boosts.length ? ` (boost: ${engine.boosts.join(', ')})` : '')]);
|
||||
rows.push(['Engine restarts (run)', `memory ${diagnostics.restarts.memory}, threads ${diagnostics.restarts.threads}`]);
|
||||
rows.push(['Inference process', proc(processes?.inference)]);
|
||||
rows.push(['Main process', proc(processes?.main)]);
|
||||
rows.push(['Available memory', processes?.available ? mb(processes.available) : '—']);
|
||||
rows.push(['Slice', slice ? `${n(slice.done)} / ${n(slice.total)} chunks` : '—']);
|
||||
if (chunks) {
|
||||
let { sizes, perDocument } = chunks;
|
||||
rows.push(['Chunks stored', `${n(sizes.count)} · ${n(sizes.tokens)} tokens · mean ${n(sizes.mean)} · median ${n(sizes.median)}`]);
|
||||
rows.push(['Chunk sizes', bucketList(sizes.buckets, sizes.count, '')]);
|
||||
rows.push(['Split sections', `${pct(sizes.splitShare)} of chunks · ${n(sizes.partsPerSplitSection, 1)} parts per split section`]);
|
||||
rows.push(['Chunks per document', `${n(perDocument.count)} documents · mean ${n(perDocument.mean, 1)} · median ${n(perDocument.median)} · max ${n(perDocument.max)}`]);
|
||||
rows.push(['Documents by chunks', bucketList(perDocument.buckets, perDocument.count, '')]);
|
||||
}
|
||||
|
||||
let box = document.getElementById('semantic-search-diagnostics');
|
||||
while (box.childElementCount < rows.length * 2) {
|
||||
box.append(document.createXULElement('label'), document.createXULElement('label'));
|
||||
}
|
||||
while (box.childElementCount > rows.length * 2) {
|
||||
box.lastElementChild.remove();
|
||||
}
|
||||
rows.forEach(([label, value], i) => {
|
||||
box.children[i * 2].value = label;
|
||||
box.children[i * 2 + 1].value = value;
|
||||
});
|
||||
},
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -355,6 +355,7 @@
|
|||
<label id="semantic-search-attachments-value"/>
|
||||
</html:div>
|
||||
</html:div>
|
||||
<html:div id="semantic-search-diagnostics"/>
|
||||
</vbox>
|
||||
</groupbox>
|
||||
</vbox>
|
||||
|
|
|
|||
|
|
@ -287,7 +287,7 @@ Zotero.Embeddings = new function () {
|
|||
// Schema version of the attached embeddings database. The tables are only
|
||||
// created when this is bumped (_setUpDB() drops and recreates everything),
|
||||
// so any schema change needs a bump.
|
||||
const _dbVersion = 4;
|
||||
const _dbVersion = 5;
|
||||
|
||||
let _dbInitPromise = null;
|
||||
let _dbHooksRegistered = false;
|
||||
|
|
@ -391,9 +391,16 @@ Zotero.Embeddings = new function () {
|
|||
+ " textCheck TEXT,\n"
|
||||
+ " sectionPart INTEGER,\n"
|
||||
+ " sectionParts INTEGER,\n"
|
||||
+ " tokens INTEGER,\n"
|
||||
+ " PRIMARY KEY (itemID, chunkIndex)\n"
|
||||
+ ")"
|
||||
);
|
||||
// Chunk-shape diagnostics read this index alone, never the rows with
|
||||
// their vectors (see Indexing._getChunkShape())
|
||||
await Zotero.DB.queryAsync(
|
||||
"CREATE INDEX embeddings.itemEmbeddings_tokens "
|
||||
+ "ON itemEmbeddings (tokens, sectionParts, sectionPart)"
|
||||
);
|
||||
// The localUserKey the vectors were built against and the model that
|
||||
// produced them
|
||||
await Zotero.DB.queryAsync(
|
||||
|
|
@ -1688,71 +1695,9 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
// gets. Restarting the engine is the only reclaim and costs about a
|
||||
// second, so past this footprint it's restarted between batches.
|
||||
const INFERENCE_MEMORY_CAP = 1.5 * 1024 * 1024 * 1024;
|
||||
// How often process usage is sampled and logged during a run
|
||||
const PROC_SAMPLE_INTERVAL = 10 * 1000;
|
||||
let _procTimer = null;
|
||||
let _procCpuTimes = new Map();
|
||||
let _procLastSample = 0;
|
||||
// The inference process's footprint at the last sample, in bytes --
|
||||
// what the between-batches restart check reads
|
||||
let _inferenceFootprint = 0;
|
||||
|
||||
// One line of process usage for the debug log -- the main and inference
|
||||
// processes' footprint and CPU (in core-fractions, so several busy
|
||||
// threads read over 100%), and the memory still available -- so a
|
||||
// submitted debug log shows what indexing cost while it ran.
|
||||
async function _sampleProcesses() {
|
||||
let info = await ChromeUtils.requestProcInfo();
|
||||
let now = Date.now();
|
||||
let elapsedNS = _procLastSample ? (now - _procLastSample) * 1e6 : 0;
|
||||
_procLastSample = now;
|
||||
let inference = info.children.find(child => child.type == 'inference');
|
||||
_inferenceFootprint = inference ? inference.memory : 0;
|
||||
let procs = [
|
||||
{ label: 'main', pid: info.pid, memory: info.memory, cpuTime: info.cpuTime }
|
||||
];
|
||||
if (inference) {
|
||||
procs.push({
|
||||
label: 'inference',
|
||||
pid: inference.pid,
|
||||
memory: inference.memory,
|
||||
cpuTime: inference.cpuTime
|
||||
});
|
||||
}
|
||||
let parts = procs.map((proc) => {
|
||||
let cpu = '';
|
||||
let prev = _procCpuTimes.get(proc.pid);
|
||||
if (prev !== undefined && elapsedNS) {
|
||||
cpu = ` cpu ${Math.round((proc.cpuTime - prev) / elapsedNS * 100)}%`;
|
||||
}
|
||||
_procCpuTimes.set(proc.pid, proc.cpuTime);
|
||||
return `${proc.label} ${(proc.memory / 1024 / 1024).toFixed(0)} MB${cpu}`;
|
||||
});
|
||||
let available = _availableMemory();
|
||||
Zotero.debug('Embeddings: ' + parts.join(', ')
|
||||
+ (available ? `, ${Math.round(available / 1024 / 1024)} MB available` : ''));
|
||||
}
|
||||
|
||||
function _startProcMonitor() {
|
||||
if (_procTimer) {
|
||||
return;
|
||||
}
|
||||
_procCpuTimes = new Map();
|
||||
_procLastSample = 0;
|
||||
_inferenceFootprint = 0;
|
||||
_procTimer = setInterval(
|
||||
() => _sampleProcesses().catch(e => Zotero.logError(e)),
|
||||
PROC_SAMPLE_INTERVAL
|
||||
);
|
||||
}
|
||||
|
||||
function _stopProcMonitor() {
|
||||
if (!_procTimer) {
|
||||
return;
|
||||
}
|
||||
clearInterval(_procTimer);
|
||||
_procTimer = null;
|
||||
}
|
||||
// Time of the process sample the last cap restart acted on, so each
|
||||
// sample triggers at most one
|
||||
let _restartedOnSample = 0;
|
||||
|
||||
// Serialize model switches so rapid preference changes don't run their
|
||||
// clear/prune/re-index steps concurrently.
|
||||
|
|
@ -1916,6 +1861,7 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
}
|
||||
return;
|
||||
}
|
||||
Zotero.Embeddings.Diagnostics.recordMemoryPressure();
|
||||
if (_tokenBudget <= DEGRADED_TOKEN_BUDGET_FLOOR) {
|
||||
return;
|
||||
}
|
||||
|
|
@ -1979,7 +1925,7 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
* @return {Boolean}
|
||||
*/
|
||||
function _hasMemoryToIndex() {
|
||||
let available = _availableMemory();
|
||||
let available = Zotero.Embeddings.Diagnostics.getAvailableMemory();
|
||||
if (available && available < MIN_AVAILABLE_MEMORY) {
|
||||
Zotero.debug(`Embeddings: only ${Math.round(available / 1024 / 1024)} MB `
|
||||
+ "available -- not indexing yet");
|
||||
|
|
@ -1988,20 +1934,6 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
return true;
|
||||
}
|
||||
|
||||
// Physical memory available right now, in bytes -- 0 when the platform
|
||||
// can't say
|
||||
function _availableMemory() {
|
||||
try {
|
||||
return Cc["@mozilla.org/ml-utils;1"]
|
||||
.getService(Ci.nsIMLUtils)
|
||||
.availablePhysicalMemory || 0;
|
||||
}
|
||||
catch (e) {
|
||||
Zotero.logError(e);
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Put an itemID on the queue its kind of work belongs to. An item the
|
||||
// cache can't type goes to the regular queue, which routes it when it
|
||||
// loads (see _drainItemQueue()).
|
||||
|
|
@ -2582,8 +2514,8 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
"INSERT INTO embeddings.itemEmbeddings "
|
||||
+ "(itemID, chunkIndex, embedding, sourceHash, "
|
||||
+ "startBlock, endBlock, startOffset, endOffset, "
|
||||
+ "textCheck, sectionPart, sectionParts) "
|
||||
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
+ "textCheck, sectionPart, sectionParts, tokens) "
|
||||
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||
[
|
||||
entry.item.id,
|
||||
chunkIndex,
|
||||
|
|
@ -2595,7 +2527,8 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
isAttachment ? chunk.endOffset ?? null : null,
|
||||
isAttachment ? Zotero.Embeddings.textCheck(chunk.text) : null,
|
||||
chunk.sectionPart ?? null,
|
||||
chunk.sectionParts ?? null
|
||||
chunk.sectionParts ?? null,
|
||||
chunk.tokens ?? null
|
||||
],
|
||||
{ debugParams: false }
|
||||
);
|
||||
|
|
@ -2860,6 +2793,7 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
// halves fulltext indexing time versus item-level ordering.
|
||||
units.sort((a, b) => a.entry.chunks[a.chunkIndex].tokens
|
||||
- b.entry.chunks[b.chunkIndex].tokens);
|
||||
Zotero.Embeddings.Diagnostics.startSlice(units.length);
|
||||
|
||||
let done = 0;
|
||||
for (let i = 0; i < units.length;) {
|
||||
|
|
@ -2881,15 +2815,24 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
}
|
||||
let batch = units.slice(i, i + count);
|
||||
i += count;
|
||||
let started = Date.now();
|
||||
let vectors = await Zotero.Embeddings.embedPassages(
|
||||
batch.map(unit => embedText(unit.entry.chunks[unit.chunkIndex]))
|
||||
);
|
||||
// The arena's only reclaim is a restart (see INFERENCE_MEMORY_CAP);
|
||||
// zeroing the footprint holds the next check until a fresh sample
|
||||
if (_inferenceFootprint > INFERENCE_MEMORY_CAP) {
|
||||
let footprintMB = Math.round(_inferenceFootprint / 1024 / 1024);
|
||||
_inferenceFootprint = 0;
|
||||
Zotero.Embeddings.Diagnostics.recordBatch({
|
||||
chunks: batch.length,
|
||||
tokens: batch.reduce((sum, unit) => sum + unit.entry.chunks[unit.chunkIndex].tokens, 0),
|
||||
longest,
|
||||
inferenceMs: Date.now() - started
|
||||
});
|
||||
// The arena's only reclaim is a restart (see INFERENCE_MEMORY_CAP)
|
||||
let sample = Zotero.Embeddings.Diagnostics.getProcessSample();
|
||||
if (sample?.inference?.memory > INFERENCE_MEMORY_CAP
|
||||
&& sample.time !== _restartedOnSample) {
|
||||
_restartedOnSample = sample.time;
|
||||
let footprintMB = Math.round(sample.inference.memory / 1024 / 1024);
|
||||
await Zotero.Embeddings.shutdownEngine({ modelChanged: false });
|
||||
Zotero.Embeddings.Diagnostics.recordRestart('memory');
|
||||
Zotero.debug(`Embeddings: inference process at ${footprintMB} MB `
|
||||
+ '-- engine restarted to release its memory');
|
||||
}
|
||||
|
|
@ -2897,6 +2840,7 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
// between batches, never mid-request
|
||||
else if (Zotero.Embeddings.engineThreadsStale()) {
|
||||
await Zotero.Embeddings.shutdownEngine({ modelChanged: false });
|
||||
Zotero.Embeddings.Diagnostics.recordRestart('threads');
|
||||
Zotero.debug('Embeddings: engine restarted to apply new thread count');
|
||||
}
|
||||
let completed = [];
|
||||
|
|
@ -3001,6 +2945,20 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
extractionProgress: _extractionProgress,
|
||||
items: _itemCounts,
|
||||
chunks: _chunkCounts,
|
||||
// Seconds until the fulltext work is embedded, at the current rate.
|
||||
// Unknown while extraction is still adding to the total.
|
||||
eta: _phase === 'extracting'
|
||||
? null
|
||||
: Zotero.Embeddings.Diagnostics.estimateSeconds(_chunkCounts.total - _chunkCounts.done),
|
||||
diagnostics: {
|
||||
...Zotero.Embeddings.Diagnostics.getStatus(),
|
||||
tokenBudget: _tokenBudget,
|
||||
engine: {
|
||||
threads: this.getEngineThreads(),
|
||||
optimalThreads: Zotero.ML.getOptimalConcurrency(),
|
||||
boosts: [..._threadBoosts]
|
||||
}
|
||||
},
|
||||
error: _lastError ? (_lastError.message || String(_lastError)) : null
|
||||
};
|
||||
};
|
||||
|
|
@ -3056,6 +3014,7 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
}
|
||||
_itemCounts = { done: await _getIndexedItemCount(), total };
|
||||
_chunkCounts = await _getChunkCounts();
|
||||
await Zotero.Embeddings.Diagnostics.refreshChunkShape();
|
||||
_lastCountRefresh = Date.now();
|
||||
_emitProgress();
|
||||
return Zotero.Embeddings.Indexing.getStatus();
|
||||
|
|
@ -3072,8 +3031,8 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
|
||||
// Progress tick for a run's inner loops: refresh the chunk counts and
|
||||
// emit, at most once per PROGRESS_EMIT_INTERVAL. The indexed item count
|
||||
// costs a scan of the index, so it's recomputed less often; the eligible
|
||||
// count comes from the last full refresh.
|
||||
// and chunk shape each cost an index scan, so they're recomputed less
|
||||
// often; the eligible count comes from the last full refresh.
|
||||
async function _tick() {
|
||||
let now = Date.now();
|
||||
if (now - _lastTick < PROGRESS_EMIT_INTERVAL) {
|
||||
|
|
@ -3085,6 +3044,7 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
if (now - _lastCountRefresh >= COUNT_REFRESH_INTERVAL) {
|
||||
_lastCountRefresh = now;
|
||||
_itemCounts = { done: await _getIndexedItemCount(), total: _itemCounts.total };
|
||||
await Zotero.Embeddings.Diagnostics.refreshChunkShape();
|
||||
}
|
||||
}
|
||||
catch (e) {
|
||||
|
|
@ -3198,7 +3158,7 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
_stopping = false;
|
||||
_lastError = null;
|
||||
_lastTick = 0;
|
||||
_startProcMonitor();
|
||||
Zotero.Embeddings.Diagnostics.startRun();
|
||||
_startIdleWatch();
|
||||
try {
|
||||
await Zotero.Embeddings.initDB();
|
||||
|
|
@ -3247,11 +3207,11 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
}
|
||||
finally {
|
||||
_stopIdleWatch();
|
||||
_stopProcMonitor();
|
||||
_indexing = false;
|
||||
_phase = 'idle';
|
||||
_downloadProgress = null;
|
||||
_extractionProgress = null;
|
||||
Zotero.Embeddings.Diagnostics.endRun();
|
||||
// A run cut short by a stop or an error still reports what's
|
||||
// stored
|
||||
try {
|
||||
|
|
@ -3327,6 +3287,272 @@ Zotero.Embeddings.Indexing = new function () {
|
|||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Pipeline diagnostics, reported through Indexing.getStatus().
|
||||
* Indexing records what happens -- engine batches, restarts, memory
|
||||
* pressure, process samples, the slice in progress -- and this turns it
|
||||
* into rates and shape summaries. Tokens are the chunker's estimates.
|
||||
*/
|
||||
Zotero.Embeddings.Diagnostics = new function () {
|
||||
// Throughput window: at least a slice, since the in-slice length sort
|
||||
// makes shorter windows swing
|
||||
const RATE_WINDOW = 120 * 1000;
|
||||
// A window this long is trusted for estimates
|
||||
const ESTABLISHED_WINDOW = 30 * 1000;
|
||||
// How often the main and inference processes are sampled during a run
|
||||
const PROC_SAMPLE_INTERVAL = 10 * 1000;
|
||||
let _samples = [];
|
||||
let _run = _newRun();
|
||||
let _slice = null;
|
||||
let _chunkShape = null;
|
||||
let _pressureEvents = 0;
|
||||
let _procTimer = null;
|
||||
let _procCpuTimes = new Map();
|
||||
let _procLastSample = 0;
|
||||
let _processSample = null;
|
||||
|
||||
function _newRun() {
|
||||
return {
|
||||
batches: 0,
|
||||
chunks: 0,
|
||||
tokens: 0,
|
||||
padded: 0,
|
||||
inferenceMs: 0,
|
||||
restarts: { memory: 0, threads: 0 }
|
||||
};
|
||||
}
|
||||
|
||||
// Reset everything scoped to one indexing run and start sampling the
|
||||
// processes
|
||||
this.startRun = function () {
|
||||
_samples = [];
|
||||
_run = _newRun();
|
||||
_pressureEvents = 0;
|
||||
_slice = null;
|
||||
if (!_procTimer) {
|
||||
_procCpuTimes = new Map();
|
||||
_procLastSample = 0;
|
||||
_procTimer = setInterval(
|
||||
() => _sampleProcesses().catch(e => Zotero.logError(e)),
|
||||
PROC_SAMPLE_INTERVAL
|
||||
);
|
||||
}
|
||||
};
|
||||
|
||||
this.endRun = function () {
|
||||
_slice = null;
|
||||
if (_procTimer) {
|
||||
clearInterval(_procTimer);
|
||||
_procTimer = null;
|
||||
}
|
||||
};
|
||||
|
||||
// Throughput over the last RATE_WINDOW, wall-clock -- so it includes
|
||||
// commits and restarts -- or null before the first batch
|
||||
function _getWindow() {
|
||||
if (!_samples.length) {
|
||||
return null;
|
||||
}
|
||||
let sum = key => _samples.reduce((total, sample) => total + sample[key], 0);
|
||||
let span = Date.now() - _samples[0].time;
|
||||
let seconds = Math.max(1, span / 1000);
|
||||
let tokens = sum('tokens');
|
||||
return {
|
||||
chunksPerSecond: sum('chunks') / seconds,
|
||||
tokensPerSecond: tokens / seconds,
|
||||
paddingEfficiency: tokens / (sum('padded') || 1),
|
||||
established: span >= ESTABLISHED_WINDOW
|
||||
};
|
||||
}
|
||||
|
||||
// Seconds to embed `remaining` chunks at the window's rate, or null when
|
||||
// there's nothing left or the window isn't established
|
||||
this.estimateSeconds = function (remaining) {
|
||||
let window = _getWindow();
|
||||
if (!window?.established || remaining <= 0) {
|
||||
return null;
|
||||
}
|
||||
return remaining / window.chunksPerSecond;
|
||||
};
|
||||
|
||||
// The last process sample: { time, available, main: { memory, cpu },
|
||||
// inference: { memory, cpu } }, with memory in bytes and CPU in
|
||||
// core-fractions (several busy threads read over 100%). Null before the
|
||||
// first sample of a run; inference is absent when no engine is up.
|
||||
this.getProcessSample = function () {
|
||||
return _processSample;
|
||||
};
|
||||
|
||||
// Physical memory available right now, in bytes -- 0 when the platform
|
||||
// can't say
|
||||
this.getAvailableMemory = function () {
|
||||
try {
|
||||
return Cc["@mozilla.org/ml-utils;1"]
|
||||
.getService(Ci.nsIMLUtils)
|
||||
.availablePhysicalMemory || 0;
|
||||
}
|
||||
catch (e) {
|
||||
Zotero.logError(e);
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
async function _sampleProcesses() {
|
||||
let info = await ChromeUtils.requestProcInfo();
|
||||
let now = Date.now();
|
||||
let elapsedNS = _procLastSample ? (now - _procLastSample) * 1e6 : 0;
|
||||
_procLastSample = now;
|
||||
let inference = info.children.find(child => child.type == 'inference');
|
||||
let procs = [
|
||||
{ label: 'main', pid: info.pid, memory: info.memory, cpuTime: info.cpuTime }
|
||||
];
|
||||
if (inference) {
|
||||
procs.push({
|
||||
label: 'inference',
|
||||
pid: inference.pid,
|
||||
memory: inference.memory,
|
||||
cpuTime: inference.cpuTime
|
||||
});
|
||||
}
|
||||
let sample = { time: now, available: Zotero.Embeddings.Diagnostics.getAvailableMemory() };
|
||||
for (let proc of procs) {
|
||||
let prev = _procCpuTimes.get(proc.pid);
|
||||
_procCpuTimes.set(proc.pid, proc.cpuTime);
|
||||
sample[proc.label] = {
|
||||
memory: proc.memory,
|
||||
cpu: prev !== undefined && elapsedNS
|
||||
? Math.round((proc.cpuTime - prev) / elapsedNS * 100)
|
||||
: null
|
||||
};
|
||||
}
|
||||
_processSample = sample;
|
||||
}
|
||||
|
||||
// A slice of `total` chunks is about to be embedded
|
||||
this.startSlice = function (total) {
|
||||
_slice = { done: 0, total };
|
||||
};
|
||||
|
||||
// Record an engine batch that finished at `time` (now by default). Padded
|
||||
// tokens are what the engine computed: every text as long as the longest.
|
||||
this.recordBatch = function ({ chunks, tokens, longest, inferenceMs, time = Date.now() }) {
|
||||
let sample = { time, chunks, tokens, padded: chunks * longest, inferenceMs };
|
||||
_samples.push(sample);
|
||||
while (_samples.length && _samples[0].time < sample.time - RATE_WINDOW) {
|
||||
_samples.shift();
|
||||
}
|
||||
_run.batches++;
|
||||
_run.chunks += chunks;
|
||||
_run.tokens += tokens;
|
||||
_run.padded += sample.padded;
|
||||
_run.inferenceMs += inferenceMs;
|
||||
if (_slice) {
|
||||
_slice.done += chunks;
|
||||
}
|
||||
};
|
||||
|
||||
// @param {String} cause - 'memory' or 'threads'
|
||||
this.recordRestart = function (cause) {
|
||||
_run.restarts[cause]++;
|
||||
};
|
||||
|
||||
this.recordMemoryPressure = function () {
|
||||
_pressureEvents++;
|
||||
};
|
||||
|
||||
// Recount the chunk shape from the database (see _getChunkShape())
|
||||
this.refreshChunkShape = async function () {
|
||||
_chunkShape = await _getChunkShape();
|
||||
};
|
||||
|
||||
// The window is wall-clock throughput; the run's inference speed counts
|
||||
// only time inside the engine, so the gap between them is overhead.
|
||||
this.getStatus = function () {
|
||||
let window = _getWindow();
|
||||
let run = null;
|
||||
if (_run.batches) {
|
||||
let seconds = Math.max(0.001, _run.inferenceMs / 1000);
|
||||
run = {
|
||||
chunksPerSecond: _run.chunks / seconds,
|
||||
tokensPerSecond: _run.tokens / seconds,
|
||||
paddingEfficiency: _run.tokens / (_run.padded || 1),
|
||||
batches: _run.batches,
|
||||
chunksPerBatch: _run.chunks / _run.batches,
|
||||
tokensPerBatch: _run.tokens / _run.batches
|
||||
};
|
||||
}
|
||||
return {
|
||||
window,
|
||||
run,
|
||||
restarts: _run.restarts,
|
||||
pressureEvents: _pressureEvents,
|
||||
processes: _processSample,
|
||||
slice: _slice,
|
||||
chunks: _chunkShape
|
||||
};
|
||||
};
|
||||
|
||||
// The shape of the stored attachment chunks -- sizes from the tokens
|
||||
// index, chunks per document from the ledger -- for judging the
|
||||
// chunker's output. Attachment rows are the ones with sectionParts;
|
||||
// other item types' rows aren't the chunker's work.
|
||||
async function _getChunkShape() {
|
||||
let { BUDGET_TOKENS, MIN_TOKENS } = Zotero.Utilities.Internal.Chunking;
|
||||
let sizeBounds = [MIN_TOKENS, BUDGET_TOKENS / 2, Math.round(BUDGET_TOKENS * 5 / 6), BUDGET_TOKENS];
|
||||
let documentBounds = [1, 11, 51, 201];
|
||||
let bucketSQL = (column, bounds) => bounds.map((bound, i) => (i
|
||||
? `SUM(${column} >= ${bounds[i - 1]} AND ${column} < ${bound}) AS b${i}`
|
||||
: `SUM(${column} < ${bound}) AS b0`
|
||||
)).concat(`SUM(${column} >= ${bounds[bounds.length - 1]}) AS b${bounds.length}`).join(', ');
|
||||
let buckets = (row, bounds) => bounds.map((bound, i) => ({
|
||||
from: i ? bounds[i - 1] : null,
|
||||
to: bound,
|
||||
count: row[`b${i}`] || 0
|
||||
})).concat({ from: bounds[bounds.length - 1], to: null, count: row[`b${bounds.length}`] || 0 });
|
||||
let median = async (sql, count) => (count
|
||||
? Zotero.DB.valueQueryAsync(sql + " LIMIT 1 OFFSET " + Math.floor(count / 2))
|
||||
: 0);
|
||||
|
||||
let sizes = await Zotero.DB.rowQueryAsync(
|
||||
"SELECT COUNT(*) AS count, COALESCE(SUM(tokens), 0) AS tokens, "
|
||||
+ bucketSQL('tokens', sizeBounds) + ", "
|
||||
+ "SUM(sectionParts > 1) AS split, "
|
||||
+ "SUM(sectionParts > 1 AND sectionPart = 1) AS splitSections, "
|
||||
+ "SUM(CASE WHEN sectionParts > 1 AND sectionPart = 1 THEN sectionParts ELSE 0 END) AS splitParts "
|
||||
+ "FROM embeddings.itemEmbeddings WHERE sectionParts IS NOT NULL"
|
||||
);
|
||||
let documents = await Zotero.DB.rowQueryAsync(
|
||||
"SELECT COUNT(*) AS count, COALESCE(SUM(chunks), 0) AS chunks, "
|
||||
+ "COALESCE(MAX(chunks), 0) AS max, " + bucketSQL('chunks', documentBounds)
|
||||
+ " FROM embeddings.itemChunkCounts"
|
||||
);
|
||||
return {
|
||||
sizes: {
|
||||
count: sizes.count,
|
||||
tokens: sizes.tokens,
|
||||
mean: sizes.count ? sizes.tokens / sizes.count : 0,
|
||||
median: await median(
|
||||
"SELECT tokens FROM embeddings.itemEmbeddings WHERE sectionParts IS NOT NULL "
|
||||
+ "ORDER BY tokens",
|
||||
sizes.count),
|
||||
buckets: buckets(sizes, sizeBounds),
|
||||
splitShare: sizes.count ? (sizes.split || 0) / sizes.count : 0,
|
||||
partsPerSplitSection: sizes.splitSections ? sizes.splitParts / sizes.splitSections : 0
|
||||
},
|
||||
perDocument: {
|
||||
count: documents.count,
|
||||
mean: documents.count ? documents.chunks / documents.count : 0,
|
||||
median: await median(
|
||||
"SELECT chunks FROM embeddings.itemChunkCounts ORDER BY chunks", documents.count),
|
||||
max: documents.max,
|
||||
buckets: buckets(documents, documentBounds)
|
||||
}
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* Zotero.Embeddings.Calibration -- how a model's scoring numbers are derived.
|
||||
*
|
||||
|
|
|
|||
|
|
@ -119,6 +119,22 @@
|
|||
}
|
||||
}
|
||||
|
||||
#semantic-search-diagnostics {
|
||||
display: grid;
|
||||
grid-template-columns: max-content 1fr;
|
||||
column-gap: 12px;
|
||||
row-gap: 2px;
|
||||
margin-top: 10px;
|
||||
|
||||
label:nth-child(odd) {
|
||||
color: var(--fill-secondary);
|
||||
}
|
||||
|
||||
label:nth-child(even) {
|
||||
white-space: normal;
|
||||
}
|
||||
}
|
||||
|
||||
#db-maintenance-options {
|
||||
display: flex;
|
||||
gap: 6px;
|
||||
|
|
|
|||
|
|
@ -1051,6 +1051,28 @@ describe("Zotero.Embeddings", function () {
|
|||
});
|
||||
});
|
||||
|
||||
describe("Diagnostics", function () {
|
||||
it("should estimate remaining time once the window spans long enough", function () {
|
||||
let diagnostics = Zotero.Embeddings.Diagnostics;
|
||||
try {
|
||||
diagnostics.startRun();
|
||||
let batch = { chunks: 10, tokens: 1000, longest: 100, inferenceMs: 100 };
|
||||
diagnostics.recordBatch(batch);
|
||||
// One batch spans no time, so the window isn't trusted yet
|
||||
assert.isNull(diagnostics.estimateSeconds(300));
|
||||
// A batch a minute ago makes the window a minute wide
|
||||
diagnostics.startRun();
|
||||
diagnostics.recordBatch({ ...batch, time: Date.now() - 60 * 1000 });
|
||||
diagnostics.recordBatch(batch);
|
||||
assert.approximately(diagnostics.estimateSeconds(300), 900, 1);
|
||||
assert.isNull(diagnostics.estimateSeconds(0));
|
||||
}
|
||||
finally {
|
||||
diagnostics.endRun();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe("Calibration", function () {
|
||||
// Stub the model rather than setting the pref: writing embeddings.model
|
||||
// kicks off a real model switch, which clears the index and the stored
|
||||
|
|
@ -1772,6 +1794,73 @@ describe("Zotero.Embeddings", function () {
|
|||
}
|
||||
});
|
||||
|
||||
it("should report pipeline diagnostics", async function () {
|
||||
this.timeout(60000);
|
||||
let item = await createDataObject('item', { title: 'Parent of measured attachment' });
|
||||
let attachment = await importPDFAttachment(item);
|
||||
let vector = new Float32Array(4).fill(0.5);
|
||||
let stubs = [
|
||||
sinon.stub(Zotero.Embeddings, 'embedPassages')
|
||||
.callsFake(async texts => texts.map(() => vector)),
|
||||
sinon.stub(Zotero.Embeddings, 'isEnabled').returns(true),
|
||||
sinon.stub(Zotero.Embeddings, 'getModelVersion').returns('test-model/1'),
|
||||
sinon.stub(Zotero.Embeddings, 'isDownloaded').resolves(true),
|
||||
sinon.stub(Zotero.Embeddings, 'download').resolves(),
|
||||
sinon.stub(Zotero.Embeddings, 'ensureCalibration').resolves(),
|
||||
sinon.stub(Zotero.Embeddings, 'getModelName').returns('bge-small-en-v1.5'),
|
||||
sinon.stub(Zotero.SDT, 'ensure').resolves(true),
|
||||
sinon.stub(Zotero.SDT, 'getSections').resolves({
|
||||
ok: true,
|
||||
sections: [
|
||||
sdtSection('', 0, ['Owls hunt at night. '.repeat(200)]),
|
||||
sdtSection('', 1, ['Hawks hunt by day. '.repeat(200)])
|
||||
]
|
||||
})
|
||||
];
|
||||
try {
|
||||
Zotero.Prefs.set('embeddings.indexFulltext', true);
|
||||
await Zotero.Embeddings.Indexing.startIndexing();
|
||||
let { diagnostics } = await Zotero.Embeddings.Indexing.refreshStatus();
|
||||
|
||||
// Every stored chunk carries its size
|
||||
assert.equal(await Zotero.DB.valueQueryAsync(
|
||||
"SELECT COUNT(*) FROM embeddings.itemEmbeddings WHERE itemID=? AND tokens IS NULL",
|
||||
attachment.id
|
||||
), 0);
|
||||
|
||||
// Rates come from the run's batches
|
||||
assert.isAbove(diagnostics.run.batches, 0);
|
||||
assert.isAbove(diagnostics.run.chunksPerSecond, 0);
|
||||
assert.isAbove(diagnostics.run.tokensPerSecond, 0);
|
||||
assert.isAbove(diagnostics.run.paddingEfficiency, 0);
|
||||
assert.isAtMost(diagnostics.run.paddingEfficiency, 1);
|
||||
assert.isAbove(diagnostics.window.chunksPerSecond, 0);
|
||||
assert.isNull(diagnostics.slice);
|
||||
assert.isAtLeast(diagnostics.engine.threads, 1);
|
||||
// Nothing left to embed, so no estimate
|
||||
assert.isNull((await Zotero.Embeddings.Indexing.refreshStatus()).eta);
|
||||
|
||||
// Chunk shape agrees with the tables
|
||||
let { sizes, perDocument } = diagnostics.chunks;
|
||||
assert.equal(sizes.count, await Zotero.DB.valueQueryAsync(
|
||||
"SELECT COUNT(*) FROM embeddings.itemEmbeddings WHERE sectionParts IS NOT NULL"
|
||||
));
|
||||
assert.equal(sizes.buckets.reduce((sum, b) => sum + b.count, 0), sizes.count);
|
||||
assert.isAbove(sizes.median, 0);
|
||||
assert.equal(perDocument.count, await Zotero.DB.valueQueryAsync(
|
||||
"SELECT COUNT(*) FROM embeddings.itemChunkCounts"
|
||||
));
|
||||
assert.equal(perDocument.buckets.reduce((sum, b) => sum + b.count, 0), perDocument.count);
|
||||
assert.isAtLeast(perDocument.max, await Zotero.DB.valueQueryAsync(
|
||||
"SELECT chunks FROM embeddings.itemChunkCounts WHERE itemID=?", attachment.id
|
||||
));
|
||||
}
|
||||
finally {
|
||||
stubs.forEach(stub => stub.restore());
|
||||
Zotero.Prefs.clear('embeddings.indexFulltext');
|
||||
}
|
||||
});
|
||||
|
||||
it("should index auxiliary chunks with words and drop bare labels", async function () {
|
||||
this.timeout(60000);
|
||||
let item = await createDataObject('item', { title: 'Parent of captioned attachment' });
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue