Replace the best-match top-K cutoff with a ranked Relevance column

Semantic similarity has no natural relevance threshold, so instead of
asking the user to pick an arbitrary result count, show every scored
item and surface the ranking directly: a Relevance column appears and
becomes the sort while a best-match search is active, and the previous
sort and columns return when it clears. The merged results are scored
in a single pass in the row provider, so ranks are global across a
multi-collection selection, child items (attachments, notes,
annotations) rank via their top-level item, and equal scores get equal
ranks that order deterministically via the secondary sort fields. Items
without a stored embedding are filtered out.

Each cell renders the score's position within the model's display range
as a bar, so relevant results read as full and the irrelevant tail
reads as empty. The ranges are provisional per-model display constants.
Sorting uses the ranks, which are also exposed to assistive technology
and as the cell tooltip. On a focused selected row the bar
switches to white so the fill doesn't vanish into the accent selection
background.
This commit is contained in:
Dan Stillman 2026-07-20 12:38:36 -04:00 • committed by Bogdan Abaev
parent bbd8d266b0
commit ddeaefdbe0
15 changed files with 477 additions and 120 deletions

View file

@ -107,6 +107,12 @@ const STUB_COLLECTION_TREE_ROW = {
clearCache: () => {}
};
// Collection tree rows can be duck-typed stand-ins (e.g. the citation
// dialog's), which implement only part of the row API
function rowIsBestMatchSearch(row) {
return typeof row.isBestMatchSearch == 'function' && row.isBestMatchSearch();
}
class CollectionViewItemTreeRowProvider extends ItemTreeRowProvider {
constructor(itemTree) {
super(itemTree);
@ -161,6 +167,87 @@ class CollectionViewItemTreeRowProvider extends ItemTreeRowProvider {
return this.collectionTreeRows[0]?.searchText.length > 0;
}
/**
* Best-match ranks for the Relevance column, computed over the merged
* result set in _refresh() while a best-match search is active
*
* @returns {Map} - treeViewID -> 1-based rank (1 = most similar)
*/
getBestMatchRanks() {
return this._bestMatchRanks || new Map();
}
/**
* Score fractions for the Relevance column's bars, computed alongside the
* ranks (see Zotero.Embeddings.getScoreFraction())
*
* @returns {Map} - treeViewID -> 0-1 fraction of the model's display range
*/
getBestMatchBarFractions() {
return this._bestMatchBarFractions || new Map();
}
/**
* The semantic stage of a best-match search: score the merged,
* deduplicated results from all selected rows against the query in a
* single call, and keep the scoreable items ranked globally across the
* selection. Child items (attachments, notes, annotations) are scored via
* their top-level item, so result sets at other levels (e.g. a saved
* search returning annotations) rank by their parent item. Equal scores
* get equal ranks, so tied rows (including a child and its parent) order
* deterministically via the secondary sort fields.
*
* @param {Zotero.Item[]} items - Merged results from all selected rows
* @return {Promise<Zotero.Item[]>} - The scoreable items
*/
async _applyBestMatch(items) {
let query = this.collectionTreeRows.find(rowIsBestMatchSearch).searchText;
// Map each item to the item whose embedding scores it
let sourceIDByItem = new Map();
for (let item of items) {
if (!(item instanceof Zotero.Item)) {
continue;
}
let source = item.isRegularItem() ? item : item.topLevelItem;
if (source) {
sourceIDByItem.set(item, source.id);
}
}
let scores;
try {
scores = await Zotero.Embeddings.scoreItemIDs(query, [...new Set(sourceIDByItem.values())]);
}
catch (e) {
// Scoring can fail while the model is still downloading or the
// index is being rebuilt -- show no results rather than an
// unranked scope
Zotero.logError(e);
this._bestMatchRanks = new Map();
return [];
}
let rankOfScore = new Map(
[...new Set(scores.values())].sort((a, b) => b - a).map((score, i) => [score, i + 1])
);
let kept = [];
let ranks = new Map();
let fractions = new Map();
for (let item of items) {
let sourceID = sourceIDByItem.get(item);
if (sourceID === undefined || !scores.has(sourceID)) {
continue;
}
kept.push(item);
ranks.set(item.treeViewID, rankOfScore.get(scores.get(sourceID)));
fractions.set(
item.treeViewID,
Zotero.Embeddings.getScoreFraction(scores.get(sourceID))
);
}
this._bestMatchRanks = ranks;
this._bestMatchBarFractions = fractions;
return kept;
}
/**
* When showing multiple libraries, group rows by library in collections-list
* order -- independent of the active sort direction
@ -390,6 +477,8 @@ class CollectionViewItemTreeRowProvider extends ItemTreeRowProvider {
try {
this.collectionTreeRows.forEach(row => row.clearCache());
this._bestMatchRanks = null;
this._bestMatchBarFractions = null;
// Get the full set of items we want to show, merged across all selected rows
let newSearchItemSet = new Set();
for (let arr of await Promise.all(this.collectionTreeRows.map(row => row.getItems()))) {
@ -397,6 +486,24 @@ class CollectionViewItemTreeRowProvider extends ItemTreeRowProvider {
newSearchItemSet.add(item);
}
}
// A selected saved search's own conditions -- where a bestMatch
// marker lives -- aren't necessarily loaded yet, since its search
// runs on a clone
// isSearch() alone isn't enough: duck-typed rows (e.g. the citation
// dialog's) report it for rows whose refs aren't searches. The ref
// check alone isn't either: Unfiled-style rows hold transient,
// unsaved searches that can't load conditions.
await Promise.all(this.collectionTreeRows
.filter(row => typeof row.isSearch == 'function' && row.isSearch()
&& row.ref instanceof Zotero.Search)
.map(row => row.ref.loadDataType('conditions')));
// Entering, refreshing within, or leaving a best-match search
// changes the effective sort of rows already in the tree (the
// forced Relevance sort comes and goes, and ranks change with the
// query), so a partial sort of just the added rows isn't enough
let bestMatchSearch = this.collectionTreeRows.some(rowIsBestMatchSearch);
let forceSortAll = options.forceSortAll || bestMatchSearch || this._wasBestMatchSearch;
this._wasBestMatchSearch = bestMatchSearch;
let newSearchItems = [...newSearchItemSet];
// Embedded-image attachments (images pasted into notes) are never shown in the
// tree, so don't let one match a search and pull in its parents
@ -434,6 +541,10 @@ class CollectionViewItemTreeRowProvider extends ItemTreeRowProvider {
|| item.isRegularItem();
});
}
// The semantic stage: one scoring pass over the merged results
if (bestMatchSearch) {
newSearchItems = await this._applyBestMatch(newSearchItems);
}
let newSearchItemIDs = new Set(newSearchItems.map(item => item.treeViewID));
// In Recently Read, the search matches parent items, but the items that were
// actually read are their child attachments. Mark those as matched too, so they
@ -554,7 +665,7 @@ class CollectionViewItemTreeRowProvider extends ItemTreeRowProvider {
// In grouped mode, always sort everything: a partial sort doesn't compare
// pre-existing rows against each other, so library grouping wouldn't be
// applied to rows carried over from the previous view
this._sort(options.forceSortAll || this._groupedByLibrary ? null : [...addedItemIDs]);
this._sort(forceSortAll || this._groupedByLibrary ? null : [...addedItemIDs]);
// Toggle all open containers closed and open to refresh child items
var t = new Date();
@ -770,7 +881,8 @@ class CollectionViewItemTreeRowProvider extends ItemTreeRowProvider {
refresh = true;
madeChanges = true;
}
// Under an active best-match quick search, handle removals with a full refresh too
// Under an active best-match quick search, handle removals with a full
// refresh too, so the remaining rows' relevance ranks are recomputed
else if (['remove', 'delete', 'trash'].includes(action)
&& collectionTreeRows.some(row => row.isBestMatchSearch())) {
this.itemTree.invalidateRowCache(ids);

View file

@ -85,3 +85,4 @@ module.exports.getCSSItemTypeIcon = function (itemType, key = 'item-type') {
module.exports['IconAttachSmall'] = props => <CSSIcon name="attachment" className="icon-16" {...props} />;
module.exports['IconTreeitemNoteSmall'] = props => <CSSIcon name="note" className="icon-16" {...props} />;
module.exports['IconRelevanceSmall'] = props => <CSSIcon name="list-number" className="icon-16" {...props} />;

View file

@ -1916,14 +1916,27 @@ var Columns = class {
let columnsSettings = this._getPrefs();
let columns = this._columns = [];
// If the passed columns already carry a sort direction, the parent has
// resolved the sorted column (e.g. the forced Relevance sort), so a
// persisted direction on another column is stale and would show a
// second sort indicator
const propsHaveSort = virtualizedTable.props.columns.some(c => c.sortDirection);
for (let column of virtualizedTable.props.columns) {
// Fixed width columns can sometimes somehow obtain a width property
// this fixes it for users that may have run into the bug
if (column.fixedWidth && typeof columnsSettings[column.dataKey] == "object") {
delete columnsSettings[column.dataKey].width;
}
// Don't load column settings for disabled columns (they are overriden to be hidden)
column = Object.assign({}, column, column.disabled ? {} : columnsSettings[column.dataKey]);
// Don't load column settings for disabled columns (they are overriden
// to be hidden) or transient ones, whose state the parent derives
let settings = (column.disabled || column.transient)
? {}
: columnsSettings[column.dataKey] || {};
if (propsHaveSort && !column.sortDirection && settings.sortDirection) {
settings = Object.assign({}, settings);
delete settings.sortDirection;
}
column = Object.assign({}, column, settings);
column.className = cx(column.className, column.dataKey, column.dataKey + this._cssSuffix,
{ 'fixed-width': column.fixedWidth });
if (column.type) {
@ -2168,7 +2181,11 @@ var Columns = class {
else {
sortedColumn = column;
if (column.sortDirection) {
column.sortDirection *= -1;
// A fixed-direction column (e.g. Relevance) can be selected
// as the sort but not reversed
if (!column.fixedSortDirection) {
column.sortDirection *= -1;
}
}
else {
column.sortDirection = column.sortReverse ? -1 : 1;

View file

@ -131,33 +131,6 @@
this._advancedButton = advancedButton;
}
// Dropdown selecting how many results the best-match mode keeps;
// shown in place of the Advanced Search button
let topKList = document.createXULElement('menulist');
topKList.id = 'zotero-tb-search-topk';
topKList.hidden = true;
document.l10n.setAttributes(topKList, 'quicksearch-semantic-topk');
let topKPopup = document.createXULElement('menupopup');
for (let n of [5, 10, 25, 50, 100]) {
let item = document.createXULElement('menuitem');
item.label = String(n);
item.value = String(n);
topKPopup.append(item);
}
topKList.append(topKPopup);
topKList.value = String(Zotero.Prefs.get('search.quicksearch-semantic-topK'));
topKList.addEventListener('command', (event) => {
// Don't trigger a quick search via the oncommand handler
event.stopPropagation();
Zotero.Prefs.set('search.quicksearch-semantic-topK', parseInt(topKList.value));
// Re-run the current search with the new top-K
if (this.value) {
this.dispatchEvent(new Event('command'));
}
});
wrapper.appendChild(topKList);
this._topKList = topKList;
this.deck = this.firstElementChild;
this.querySelector('.advanced-collapse-button').addEventListener('command', (event) => {
@ -254,13 +227,11 @@
.setAttribute('checked', 'true');
document.l10n.setAttributes(this.searchTextbox.inputField, "quicksearch-input", { placeholder: this._searchModes[mode] });
// Advanced Search doesn't apply to semantic search, so swap its
// button for the similarity result-count dropdown
let isSimilarity = mode === 'bestMatch';
// A best-match search can't be converted into Advanced Search
// conditions, so hide the button in best-match mode
if (this._advancedButton) {
this._advancedButton.hidden = isSimilarity;
this._advancedButton.hidden = mode === 'bestMatch';
}
this._topKList.hidden = !isSimilarity;
let advancedSearchDeck = document.getElementById('zotero-advanced-search-pane-deck');
if (advancedSearchDeck) {

View file

@ -137,6 +137,17 @@ class ItemTreeRowProvider {
return !!row.sortChildren;
}
/**
* Best-match ranks for the Relevance column while a best-match quick
* search is active. Overridden by row providers that support best-match
* searches.
*
* @returns {Map} - treeViewID -> 1-based rank (1 = most similar)
*/
getBestMatchRanks() {
return new Map();
}
get includeTrashed() {
return this._includeTrashed;
}
@ -634,6 +645,13 @@ class ItemTreeRowProvider {
val = row.ref.getItemLastRead() || '';
break;
case 'relevance':
// Negated rank under a descending sort, so the most similar items
// (the fullest bars) come first and rows without a rank (e.g.
// child rows) sort last
val = -(this.getBestMatchRanks().get(row.id) ?? Number.MAX_SAFE_INTEGER);
break;
case 'addedBy':
val = row.ref.createdByUserID
? Zotero.Users.getName(row.ref.createdByUserID) : '';
@ -717,6 +735,11 @@ class ItemTreeRowProvider {
return Zotero.Utilities.Item.compareCallNumbers(fieldA, fieldB);
}
// Ranks are numbers, so a string comparison would misorder them
if (sortField == 'relevance') {
return fieldA - fieldB;
}
return this._sortCollation.compareString(1, String(fieldA), String(fieldB));
}
}
@ -1079,6 +1102,18 @@ var ItemTree = class ItemTree extends LibraryTree {
return true;
}
/**
* Whether a best-match quick search is active in any selected collection
* tree row, meaning the Relevance column is shown and sorted on
*/
_isBestMatchSearchActive() {
// Collection tree rows can be duck-typed stand-ins (e.g. the citation
// dialog's) that implement only part of the row API
return !!this.collectionTreeRows?.some(
row => typeof row.isBestMatchSearch == 'function' && row.isBestMatchSearch()
);
}
get hasDependOnChildrenColumn() {
return this._hasDependOnChildrenColumn;
}
@ -1287,6 +1322,17 @@ var ItemTree = class ItemTree extends LibraryTree {
rows.forEach(row => this.tree.invalidateRow(row));
}
// A refresh can change the derived column set (e.g. the forced
// Relevance column while a best-match search is active), and the
// header only picks that up through a render
this._getColumns();
if (this.tree && this._renderedColumnsId !== this._columnsId) {
await new Promise(resolve => this.forceUpdate(resolve));
// The rows above were painted with the previous column set, and the
// render only rebuilds the header
this.tree.invalidate();
}
const itemsViewInActiveWindow = Zotero.getActiveZoteroPane()?.itemsView == this;
const prioritizeRestore = !(options.selectInActiveWindow && itemsViewInActiveWindow);
const ensureVisible = options.restoreScroll ? false : options.ensureRowsAreVisible;
@ -1437,6 +1483,9 @@ var ItemTree = class ItemTree extends LibraryTree {
const showMessage = !!this._itemsPaneMessage;
const itemsPaneMessage = this._renderItemsPaneMessage(showMessage);
let columns = this._getColumns();
// The columns the header currently shows, for handleRowModelUpdate()
this._renderedColumnsId = this._columnsId;
let virtualizedTable = React.createElement(VirtualizedTree,
{
getRowCount: () => this.rowProvider.getRowCount(),
@ -1448,7 +1497,7 @@ var ItemTree = class ItemTree extends LibraryTree {
key: "virtualized-table",
showHeader: true,
columns: this._getColumns(),
columns,
onColumnPickerMenu: this._displayColumnPickerMenu.bind(this),
onColumnSort: this.isSortable ? this._handleColumnSort : null,
getColumnPrefs: this._getColumnPrefs.bind(this),
@ -1795,7 +1844,12 @@ var ItemTree = class ItemTree extends LibraryTree {
if (row.ref.isFeedItem) {
return this.getCellText(index, 'title');
}
return this.getCellText(index, this.getSortField());
let field = this.getSortField();
// Rank numbers aren't useful for find-as-you-type
if (field == 'relevance') {
field = 'title';
}
return this.getCellText(index, field);
}
/**
@ -1861,6 +1915,9 @@ var ItemTree = class ItemTree extends LibraryTree {
}
getSortField() {
// Re-derive the columns first, since a state change (e.g. a best-match
// search starting or clearing) can move the sorted column
this._getColumns();
var column = this._sortedColumn;
if (!column) {
column = this._getColumns().find(col => !col.hidden);
@ -2402,6 +2459,7 @@ var ItemTree = class ItemTree extends LibraryTree {
}
}
row.numNotes = treeRow.numNotes() || "";
row.relevance = this.rowProvider.getBestMatchRanks().get(itemID) || "";
row.feed = (treeRow.ref.isFeedItem && Zotero.Feeds.get(treeRow.ref.libraryID).name) || "";
row.lastRead = row.isItem ? treeRow.ref.getItemLastRead() : "";
row.addedBy = row.isItem ? treeRow.getAddedBy() : "";
@ -2576,11 +2634,22 @@ var ItemTree = class ItemTree extends LibraryTree {
}
_getColumns() {
const prefKey = this.id + '-' + this.viewType;
// Include the best-match-search state in the cache key, so a search
// starting or clearing rebuilds the columns with or without the forced
// Relevance column below
const bestMatchSearch = this._isBestMatchSearchActive();
const prefKey = this.id + '-' + this.viewType
+ (bestMatchSearch ? '-bestMatchSearch' : '');
if (this._columnsId == prefKey) {
return this._columns;
}
// The Relevance sort is forced only while a best-match search is
// active, so don't carry it into a rebuild without one
if (!bestMatchSearch && this._sortedColumn?.dataKey == 'relevance') {
this._sortedColumn = null;
}
this._columnsId = prefKey;
this._columns = [];
@ -2599,10 +2668,18 @@ var ItemTree = class ItemTree extends LibraryTree {
else if (column.disabledIn && this._matchesViewType(column.disabledIn)) {
columnDisabled = true;
}
const columnSettings = columnsSettings[column.dataKey];
// The Relevance column is derived entirely from the
// best-match-search state below, so ignore anything persisted
// for it (e.g. from a column resize while a search was active)
const columnSettings = column.dataKey == 'relevance'
? null
: columnsSettings[column.dataKey];
// Also includes a `hidden` pref and overrides the above if available
column = Object.assign({}, column, columnSettings || {});
if (column.dataKey == 'relevance') {
column.hidden = true;
}
// If column does not have an "ordinal" field it means it
// is newly added
if (!("ordinal" in column)) {
@ -2653,6 +2730,26 @@ var ItemTree = class ItemTree extends LibraryTree {
}
}
// While a best-match search is active, show the Relevance column and
// sort by it, most similar first. The forced state lives only on this
// rebuilt column set, so the regular columns and sort come back when
// the search clears.
if (bestMatchSearch) {
let col = this._columns.find(c => c.dataKey === 'relevance');
if (col) {
// The forced sort replaces the persisted one, whose column
// would otherwise keep showing its sort indicator
for (let other of this._columns) {
if (other !== col) {
delete other.sortDirection;
}
}
col.hidden = false;
col.sortDirection = -1;
this._sortedColumn = col;
}
}
let sortedColumns = this._columns.sort((a, b) => a.ordinal - b.ordinal);
// If no column has an explicit sort direction (e.g., a fresh profile that

View file

@ -38,6 +38,8 @@ const Icons = require('components/icons');
* @property {string[]} [defaultIn] - Types of collectionTreeRow the column is default in. See itemTree.js#_matchesViewType()
* @property {boolean} [dependsOnChildren=false] - Set to true if the column depends on child item data (e.g. numNotes, lastRead)
* @property {boolean} [sortReverse=false] - Default: false. Set to true to reverse the sort order
* @property {boolean} [fixedSortDirection=false] - Default: false. Set to true to prevent clicks from reversing the sort direction
* @property {boolean} [transient=false] - Default: false. Set to true for columns whose visibility and sort are derived at runtime, so persisted settings are never applied
* @property {number} [flex=1] - Default: 1. When the column is added to the tree how much space it should occupy as a flex ratio
* @property {string} [width] - A column width instead of flex ratio. See above.
* @property {boolean} [fixedWidth] - Default: false. Set to true to disable column resizing
@ -374,6 +376,45 @@ const COLUMNS = [
staticWidth: true,
zoteroPersist: ["width", "hidden", "sortDirection"]
},
{
dataKey: "relevance",
label: "items-column-relevance",
// Shown and sorted on automatically while a best-match search is
// active (see ItemTree#_getColumns()); not user-toggleable or persisted
showInColumnPicker: false,
iconLabel: <Icons.IconRelevanceSmall />,
width: "60",
staticWidth: true,
// Most similar first; "least similar first" isn't a useful view
sortReverse: true,
fixedSortDirection: true,
// Visibility and sort are derived from the best-match-search state, so
// persisted settings are never applied
transient: true,
zoteroPersist: [],
// A bar showing the score's fraction of the model's display range.
// `data` is the rank, which drives the sort.
renderCell(index, data, column, isFirstColumn, doc) {
let cell = doc.createElement('span');
cell.className = `cell ${column.className}`;
let fraction = this.rowProvider.getBestMatchBarFractions()
.get(this.getRow(index).id);
if (fraction !== undefined) {
let bar = doc.createElement('span');
bar.className = 'relevance-bar';
let fill = doc.createElement('span');
fill.className = 'relevance-bar-fill';
fill.style.width = Math.round(fraction * 100) + '%';
bar.append(fill);
cell.append(bar);
// The rank reaches assistive technology via the row label; show
// it visually as a tooltip
doc.l10n.formatValue('items-column-relevance-rank', { rank: data })
.then(label => cell.title = label);
}
return cell;
}
},
{
dataKey: "addedBy",
enabledIn: ["group"],

View file

@ -377,20 +377,7 @@ Zotero.CollectionTreeRow.prototype.getSearchResults = async function (asTempTabl
if (!this._cachedResults) {
let s = await this.getSearchObject();
try {
let results = await s.search();
// Similarity (semantic) quick search: re-rank the scoped results by
// similarity to the query and keep the top K (the
// search.quicksearch-semantic-topK pref). Assign the cache only once
// ranking succeeds, so a ranking failure (e.g. model not yet
// downloaded) doesn't leave the full unranked scope cached as the
// search result.
if (this.isBestMatchSearch()) {
results = await Zotero.Embeddings.rankItemIDs(
this.searchText, results,
{ limit: Zotero.Prefs.get('search.quicksearch-semantic-topK') }
);
}
this._cachedResults = results;
this._cachedResults = await s.search();
}
catch (e) {
Zotero.logError(e);
@ -634,22 +621,16 @@ Zotero.CollectionTreeRow.prototype.clearCache = function () {
Zotero.CollectionTreeRow.prototype.setSearch = function (searchText, mode = null) {
// Callers usually pass no mode (the active mode lives in the pref), so compare
// the effective mode from the last call: switching modes with unchanged text
// has to trigger a re-run -- as does changing the similarity top-K, which
// lives in its own pref. With no search text, neither matters.
// has to trigger a re-run. With no search text, the mode doesn't matter.
let effectiveMode = mode || Zotero.Prefs.get('search.quicksearch-mode');
let topK = effectiveMode === 'bestMatch'
? Zotero.Prefs.get('search.quicksearch-semantic-topK')
: null;
if (this.searchText === searchText
&& (!searchText
|| (this._effectiveSearchMode === effectiveMode && this._searchTopK === topK))) {
&& (!searchText || this._effectiveSearchMode === effectiveMode)) {
return false;
}
this.clearCache();
this.searchText = searchText;
this.searchMode = mode;
this._effectiveSearchMode = effectiveMode;
this._searchTopK = topK;
return true;
};
@ -722,7 +703,9 @@ Zotero.CollectionTreeRow.prototype.isSearchMode = function () {
/**
* Whether an active quick search on this row is in the semantic similarity
* mode -- i.e. the results are a ranked top-K set rather than a plain filter
* mode -- i.e. the results are a scored set that the items list orders by
* the Relevance column rather than a plain filter. The scoring itself is
* applied to the merged results by the items view's row provider.
*/
Zotero.CollectionTreeRow.prototype.isBestMatchSearch = function () {
return !!this.searchText && !this.advancedSearch

View file

@ -27,7 +27,7 @@
*
* Zotero.Embeddings -- the embedding engine and its public face: model
* config + download, the inference worker (bundled transformers.js + ONNX
* Runtime, run off the main thread), embed*(), and rankItemIDs() for the
* Runtime, run off the main thread), embed*(), and scoreItemIDs() for the
* search path.
*
* Zotero.Embeddings.Indexing -- everything that decides what gets embedded
@ -47,6 +47,12 @@ Zotero.Embeddings = new function () {
// bge prepends a retrieval instruction to queries; passages get none.
queryPrefix: 'Represent this sentence for searching relevant passages: ',
passagePrefix: '',
// Raw cosine scores cluster in a model-specific band; this maps that
// band onto the Relevance column's 0-1 bar (see getScoreFraction()).
// Display-only, so retuning it doesn't require a revision bump.
// Fitted to observed distributions: irrelevant mass ~0.44-0.50,
// strong matches ~0.65-0.75.
displayScoreRange: [0.5, 0.75],
l10nID: 'preferences-advanced-semantic-search-english',
files: [
'config.json',
@ -63,6 +69,7 @@ Zotero.Embeddings = new function () {
pooling: 'mean',
queryPrefix: 'query: ',
passagePrefix: 'passage: ',
displayScoreRange: [0.78, 0.92],
l10nID: 'preferences-advanced-semantic-search-multilingual',
files: [
'config.json',
@ -648,6 +655,24 @@ Zotero.Embeddings = new function () {
}
/**
* Map a raw similarity score onto the active model's display range, for
* the Relevance column's bar. The ranges are empirical per-model
* constants (see displayScoreRange in MODELS): scores at or below the
* floor render as an empty bar, at or above the ceiling as a full one.
*
* @param {Number} score
* @return {Number} - 0-1
*/
this.getScoreFraction = function (score) {
let model = MODELS[this.getModelName()];
if (!model) {
return 0;
}
let [min, max] = model.displayScoreRange;
return Math.min(1, Math.max(0, (score - min) / (max - min)));
};
// mozStorage returns a BLOB as an array of byte values; reinterpret those
// bytes as the stored Float32 embedding vector.
function _blobToVector(blob) {
@ -656,20 +681,19 @@ Zotero.Embeddings = new function () {
}
/**
* Rank a given set of items by similarity to a query, returning the most
* similar item IDs (highest first). Items without a stored embedding are
* dropped. Used to apply semantic ranking within an existing result scope
* (e.g. the current collection) rather than the whole library.
* Score a given set of items by similarity to a query. Items without a
* stored embedding aren't scored. Used to apply semantic ranking within an
* existing result scope (e.g. the current collection) rather than the
* whole library.
*
* @param {String} queryText
* @param {Number[]} itemIDs - Candidate item IDs to rank
* @param {Object} [options]
* @param {Number} [options.limit] - Keep only this many top results
* @return {Promise<Number[]>} - Ranked item IDs, most similar first
* @param {Number[]} itemIDs - Candidate item IDs to score
* @return {Promise<Map>} - itemID -> similarity score (higher is more similar)
*/
this.rankItemIDs = async function (queryText, itemIDs, { limit } = {}) {
this.scoreItemIDs = async function (queryText, itemIDs) {
let scores = new Map();
if (!itemIDs.length || !this.isEnabled()) {
return [];
return scores;
}
await this.initDB();
let query = await this.embedQuery(queryText);
@ -677,7 +701,6 @@ Zotero.Embeddings = new function () {
// Load embeddings for the candidates in chunks (avoids the SQLite bound-
// parameter limit for large collections), scoring each as we go.
let scored = [];
let chunkSize = 500;
for (let i = 0; i < itemIDs.length; i += chunkSize) {
let chunk = itemIDs.slice(i, i + chunkSize);
@ -692,15 +715,10 @@ Zotero.Embeddings = new function () {
for (let d = 0; d < dim; d++) {
dot += query[d] * vec[d];
}
scored.push({ itemID: row.itemID, score: dot });
scores.set(row.itemID, dot);
}
}
scored.sort((a, b) => b.score - a.score);
if (limit !== undefined) {
scored = scored.slice(0, limit);
}
return scored.map(s => s.itemID);
return scores;
};
/**

View file

@ -461,6 +461,8 @@ items-table-cell-notes =
items-column-added-by = Added By
items-column-modified-by = Modified By
items-column-last-read = Last Read
items-column-relevance = Relevance
items-column-relevance-rank = Rank { $rank }
report-error =
.label = Report Error…
@ -845,8 +847,6 @@ quicksearch-input =
.aria-label = Quick Search
.placeholder = { $placeholder }
.aria-description = { $placeholder }
quicksearch-semantic-topk =
.aria-label = Number of best-match search results
quickSearch-mode-similarity = Similarity
advanced-search = Advanced Search

View file

@ -108,8 +108,6 @@ pref("extensions.zotero.keys.toggleRead", "`");
pref("extensions.zotero.keys.showTabsMenu", ";");
pref("extensions.zotero.search.quicksearch-mode", "fields");
// Number of results the "Similarity" (semantic) quick search mode keeps
pref("extensions.zotero.search.quicksearch-semantic-topK", 25);
// Fulltext indexing
pref("extensions.zotero.fulltext.textMaxLength", 500000);

View file

@ -40,6 +40,7 @@ $-icons: (
attachment: 16,
chevron-6: 8,
filter: 16,
list-number: 16,
note: 16,
x-8: 16,
play: 16,

View file

@ -50,6 +50,7 @@
text-align: center;
}
.cell:first-child {
&::before {
content: "";
@ -206,8 +207,47 @@
background: transparent;
}
}
// The similarity bar: the score's fraction of the model's display range
.cell.relevance {
align-items: center;
.relevance-bar {
display: block;
flex: 1;
height: 6px;
border-radius: 3px;
background: var(--fill-quarternary);
overflow: hidden;
.relevance-bar-fill {
display: block;
height: 100%;
background: var(--accent-blue);
}
// On a focused selected row the accent fill would vanish into
// the accent selection background, so switch to white like
// .attachment-progress
@include state(".row.selected") {
background: #ffffff33;
.relevance-bar-fill {
background: var(--accent-white);
}
@include state(".virtualized-table:not(:focus-within)") {
background: var(--fill-quinary);
.relevance-bar-fill {
background: var(--accent-blue);
}
}
}
}
}
}
.cell.hasAttachment {
height: 100%;
// Don't show ellipsis

View file

@ -118,36 +118,6 @@ quick-search-textbox {
&:has(~ #zotero-tb-search-advanced-button) {
padding-inline-end: 30px;
}
// In similarity mode the Top-K dropdown replaces the Advanced Search button
// and is wider, so push the input text and the clear icon further in
// (must come after the rule above -- the hidden Advanced Search button still
// matches its :has() selector)
&:has(~ #zotero-tb-search-topk:not([hidden])) {
padding-inline-end: 62px;
}
}
// Top-K dropdown for the similarity quick search mode: overlay the end of the
// search field, where the Advanced Search button it replaces normally sits
#zotero-tb-search #zotero-tb-search-topk {
--topk-width: 52px;
position: relative;
width: var(--topk-width);
min-width: var(--topk-width);
height: 22px;
min-height: 22px;
margin: 0;
margin-inline-start: calc(-1 * var(--topk-width) - 5px);
margin-inline-end: 2px;
padding-inline: 2px;
align-self: center;
z-index: 2;
&::part(label) {
justify-content: center;
font-weight: normal;
}
}
// Match the specificity of the #zotero-items-toolbar toolbarbutton rules

View file

@ -189,11 +189,11 @@ describe("CollectionViewItemTree", function () {
let col = await createDataObject('collection');
let item = await createDataObject('item', { title: "test", collections: [col.id] });
await zp.collectionsView.selectCollection(col.id);
quicksearch.value = "test";
quicksearch.doCommand();
await itemsView._refreshPromise;
await zp.itemsView.selectItems([item.id]);
item.removeFromCollection(col.id);
await item.saveTx();
@ -202,6 +202,94 @@ describe("CollectionViewItemTree", function () {
assert.equal(quicksearch.value, "test");
});
describe("in best-match mode", function () {
var stubs = [];
beforeEach(function () {
stubs.push(sinon.stub(Zotero.Embeddings, 'isEnabled').returns(true));
stubs.push(sinon.stub(Zotero.Embeddings, 'getScoreFraction').callsFake(score => score));
Zotero.Prefs.set('search.quicksearch-mode', 'bestMatch');
});
afterEach(async function () {
stubs.forEach(stub => stub.restore());
stubs = [];
Zotero.Prefs.set('search.quicksearch-mode', 'fields');
await zp.itemsView.setFilter('search', '');
});
it("should show scored items ordered by a forced Relevance sort and restore the sort when cleared", async function () {
let col = await createDataObject('collection');
let itemA = await createDataObject('item', { title: "A", collections: [col.id] });
let itemB = await createDataObject('item', { title: "B", collections: [col.id] });
let itemC = await createDataObject('item', { title: "C", collections: [col.id] });
stubs.push(sinon.stub(Zotero.Embeddings, 'scoreItemIDs').callsFake(async (query, itemIDs) => {
let scores = new Map();
if (itemIDs.includes(itemA.id)) {
scores.set(itemA.id, 0.5);
}
if (itemIDs.includes(itemB.id)) {
scores.set(itemB.id, 0.9);
}
return scores;
}));
await select(win, col);
itemsView = zp.itemsView;
let defaultSortField = itemsView.getSortField();
await itemsView.setFilter('search', 'some query');
// Only the scored items, most similar first, despite title order
assert.deepEqual(itemsView._rows.map(row => row.id), [itemB.id, itemA.id]);
assert.equal(itemsView.getSortField(), 'relevance');
// The Relevance cells show the ranks
assert.equal(itemsView.getCellText(0, 'relevance'), 1);
assert.equal(itemsView.getCellText(1, 'relevance'), 2);
// Score fractions for the bars
assert.equal(itemsView.rowProvider.getBestMatchBarFractions().get(itemB.id), 0.9);
assert.equal(itemsView.rowProvider.getBestMatchBarFractions().get(itemA.id), 0.5);
assert.isFalse(itemsView._getColumns().find(c => c.dataKey == 'relevance').hidden);
// Clearing the search restores the previous sort and columns
await itemsView.setFilter('search', '');
assert.equal(itemsView.getSortField(), defaultSortField);
assert.isTrue(itemsView._getColumns().find(c => c.dataKey == 'relevance').hidden);
assert.deepEqual(
itemsView._rows.map(row => row.id),
[itemA.id, itemB.id, itemC.id]
);
});
it("should score once across a multi-collection selection", async function () {
let col1 = await createDataObject('collection');
let col2 = await createDataObject('collection');
let shared = await createDataObject('item', { collections: [col1.id, col2.id] });
let other = await createDataObject('item', { collections: [col2.id] });
let scoreStub = sinon.stub(Zotero.Embeddings, 'scoreItemIDs').callsFake(
async (query, itemIDs) => new Map(itemIDs.map(id => [id, id == shared.id ? 0.9 : 0.5]))
);
stubs.push(scoreStub);
await cv.selectByID("C" + col1.id);
await waitForItemsLoad(win);
cv.selection.toggleSelect(cv.getRowIndexByID("C" + col2.id));
await zp.onCollectionSelected();
await zp.itemsView.waitForLoad();
itemsView = zp.itemsView;
await itemsView.setFilter('search', 'some query');
// One scoring call for the whole selection, with the shared item deduplicated
assert.equal(scoreStub.callCount, 1);
assert.sameMembers(scoreStub.firstCall.args[1], [shared.id, other.id]);
assert.deepEqual(
itemsView._rows.filter(row => row.type == 'item').map(row => row.id),
[shared.id, other.id]
);
});
});
it("should expand parent item and attachment for an annotation match", async function () {
Zotero.Prefs.set("hideContextAnnotationRows", false);

View file

@ -24,6 +24,26 @@ describe("Zotero.Embeddings", function () {
});
});
describe("#getScoreFraction()", function () {
it("should clamp scores into the active model's display range", function () {
// bge-small-en-v1.5's displayScoreRange is [0.5, 0.75]
let stub = sinon.stub(Zotero.Embeddings, 'getModelName').returns('bge-small-en-v1.5');
try {
assert.equal(Zotero.Embeddings.getScoreFraction(0.4), 0);
assert.equal(Zotero.Embeddings.getScoreFraction(0.5), 0);
assert.approximately(Zotero.Embeddings.getScoreFraction(0.625), 0.5, 0.001);
assert.equal(Zotero.Embeddings.getScoreFraction(0.75), 1);
assert.equal(Zotero.Embeddings.getScoreFraction(0.99), 1);
// No known model -> empty bar
stub.returns('');
assert.equal(Zotero.Embeddings.getScoreFraction(0.9), 0);
}
finally {
stub.restore();
}
});
});
describe("Indexing", function () {
it("should remove a deleted item's embedding", async function () {
await Zotero.Embeddings.initDB();