mirror of
https://github.com/zotero/zotero.git
synced 2026-10-08 03:08:19 +00:00
add proximity criterion to lexical fulltext search
Drop the term cutoff past which lexical search skipped attachment
fulltext. It kept one rare word from carrying a long document to the
top, but it also made multi-word identifiers found only in the
fulltext ("APOB monitoring report") unreachable in hybrid mode.
Instead, a fulltext document matches only where enough of the query's
scoring terms (75%, rounded up, so all of them up to three) occur
within 200 tokens of each other (FTS5 NEAR). Item text keeps OR
membership: titles, abstracts and annotations are short already.
It is meant to counter the situation where a match on a single word
that happens to be very rare would push the item to the very top
of the ranking for a long query, outranking items that have
all other terms.
This commit is contained in:
parent
4427be9a5c
commit
267e779c8f
4 changed files with 201 additions and 239 deletions
|
|
@ -29,27 +29,15 @@
|
|||
* scores; when a semantic model is enabled (Zotero.Embeddings), both engines
|
||||
* score and their rankings are fused with Reciprocal Rank Fusion, so an item
|
||||
* can match by its words, by its meaning, or -- ranking highest -- by both.
|
||||
* What text the lexical engine reads depends on the query's shape (see
|
||||
* KEYWORD_QUERY_TERMS): everything for a keyword lookup, the items' own
|
||||
* text for longer queries. The facade owns everything a consumer would
|
||||
* otherwise need engine knowledge for: what counts as an empty query, how
|
||||
* results map onto the relevance bar, and what the failure modes are.
|
||||
* The facade owns everything a consumer would otherwise need engine
|
||||
* knowledge for: what counts as an empty query, how results map onto the
|
||||
* relevance bar, and what the failure modes are.
|
||||
*/
|
||||
Zotero.BestMatch = new function () {
|
||||
// The constant in a Reciprocal Rank Fusion contribution, 1 / (RRF_K +
|
||||
// rank): high enough that a handful of rank positions in one engine
|
||||
// can't drown out the other engine's opinion entirely
|
||||
const RRF_K = 60;
|
||||
// A query of this many scoring terms or fewer is a keyword lookup, and
|
||||
// the lexical engine searches everything for it -- exact presence is the
|
||||
// intent, wherever the words appear. A longer query states an intent the
|
||||
// words only gesture at: there the lexical engine searches the items'
|
||||
// own text (titles, abstracts, notes), where saying most of the query
|
||||
// still means answering it, and leaves the attachment fulltext -- where
|
||||
// BM25 can carry a document to the top on one rare word said often -- to
|
||||
// the model, which reads the query whole. A fully quoted query asks for
|
||||
// its literal words outright and searches everything at any length.
|
||||
const KEYWORD_QUERY_TERMS = 2;
|
||||
// How a passage's two kinds of evidence weigh against each other. The
|
||||
// model's reading leads: it is the calibrated signal, and it chose which
|
||||
// passages are worth showing. Saying the query's own words lifts a
|
||||
|
|
@ -241,20 +229,11 @@ Zotero.BestMatch = new function () {
|
|||
matches: { lexical: new Set(scores.keys()), semantic: new Set() }
|
||||
};
|
||||
}
|
||||
// What the lexical engine reads for this query (see
|
||||
// KEYWORD_QUERY_TERMS): everything for a keyword lookup or a fully
|
||||
// quoted query, the items' own text for anything longer
|
||||
let allSources = Zotero.Lexical.isFullyQuotedQuery(queryText)
|
||||
|| (await Zotero.Lexical.getScoringTermCount(queryText))
|
||||
<= KEYWORD_QUERY_TERMS;
|
||||
// Both engines score the same candidates concurrently. allSettled
|
||||
// rather than all, so one engine's failure still leaves the
|
||||
// other's rejection observed rather than unhandled
|
||||
let [lexical, semantic] = await Promise.allSettled([
|
||||
Zotero.Lexical.scoreItemIDs(queryText, itemIDs, {
|
||||
...options,
|
||||
sources: allSources ? undefined : ['itemText']
|
||||
}),
|
||||
Zotero.Lexical.scoreItemIDs(queryText, itemIDs, options),
|
||||
Zotero.Embeddings.scoreItemIDs(queryText, itemIDs, options)
|
||||
]);
|
||||
if (lexical.status == 'rejected') {
|
||||
|
|
@ -264,12 +243,7 @@ Zotero.BestMatch = new function () {
|
|||
if (semantic.reason instanceof Zotero.Embeddings.IndexNotReadyError) {
|
||||
Zotero.debug("Semantic index not ready -- ranking lexically: "
|
||||
+ semantic.reason.message);
|
||||
// With no model to read the fulltext, the lexical engine
|
||||
// covers all of it, whatever the query's length
|
||||
let scores = _nearTop(allSources
|
||||
? lexical.value
|
||||
: await Zotero.Lexical.scoreItemIDs(queryText, itemIDs, options),
|
||||
share => share);
|
||||
let scores = _nearTop(lexical.value, share => share);
|
||||
return {
|
||||
scores,
|
||||
matches: {
|
||||
|
|
|
|||
|
|
@ -26,42 +26,28 @@
|
|||
/**
|
||||
* Zotero.Lexical -- ranked lexical search over the library's own text.
|
||||
*
|
||||
* Scores how well a text answers a query rather than whether it contains every
|
||||
* word of it: any term can match, and the score accumulates the evidence, so a
|
||||
* document about owl migration in Norway still scores for "owl migration in the
|
||||
* united states" -- below the documents that cover all of it.
|
||||
* Scores how well a text answers a query: any term can match and the
|
||||
* evidence adds up, so a document missing a word still ranks, below one
|
||||
* that has them all.
|
||||
*
|
||||
* The ranking is FTS5's own BM25, asked one question per index rather than one
|
||||
* per query term. parseQuery() breaks a query into terms, buildExpression()
|
||||
* assembles them into a single OR expression, and the index scores every
|
||||
* document against it in one pass. BM25 already weighs a term by how rare it is
|
||||
* in the corpus it indexes, saturates repetition, and discounts long documents,
|
||||
* so a term filling half the library moves a score barely at all -- no
|
||||
* stoplist, no weights of our own, nothing curated by hand.
|
||||
* The ranking is FTS5's BM25. parseQuery() breaks the query into terms and
|
||||
* buildExpression() joins them into one OR expression, which each index
|
||||
* scores in a single pass. BM25 weighs a term by its rarity in the corpus,
|
||||
* saturates repetition and discounts long documents, so there is no stoplist
|
||||
* and no hand-tuned term weights. On top of it:
|
||||
*
|
||||
* Two things the query asks for that a bag of words wouldn't:
|
||||
* - Consecutive query words are added as phrase terms, so adjacent words
|
||||
* earn above scattered ones.
|
||||
* - The item-text index is scored with per-column weights, so a title match
|
||||
* counts for more than one in a note.
|
||||
* - An attachment's fulltext counts only where the query's terms occur
|
||||
* close together (see buildProximityExpression()); an item's own text is
|
||||
* short enough to count as is.
|
||||
*
|
||||
* - Adjacency. Consecutive query words are added to the expression as phrase
|
||||
* terms, so a text saying "special education" earns above one with the words
|
||||
* scattered. A phrase is rarer than either of its words, so BM25 would let it
|
||||
* decide the ranking by itself; the word terms are repeated
|
||||
* ADJACENCY_REPETITION times to hold it to a share of the score.
|
||||
* - Where the words landed. A title names a work and an abstract summarizes it,
|
||||
* so the item-text index is scored with per-column weights (see
|
||||
* COLUMN_WEIGHTS), which is BM25's own mechanism for the same idea.
|
||||
*
|
||||
* Raw BM25 is unbounded and its scale shifts with the query, so it can't be
|
||||
* shown or compared as it stands. Every score is divided by the most the
|
||||
* expression could earn, which BM25's shape makes calculable from the terms'
|
||||
* inverse document frequencies alone (see _getCeiling()). What's left is a 0-1
|
||||
* reading of how strongly a document carries the query, weighted toward its
|
||||
* rarer terms: a document missing a term forfeits that term's share, and the
|
||||
* top of the range belongs to a text about nothing else.
|
||||
*
|
||||
* Scores from the two indexes are comparable because both are that same
|
||||
* fraction, and because BM25 normalizes each by its own corpus's typical
|
||||
* document length -- the reason a one-line title and a 400-page PDF can be read
|
||||
* on one scale at all.
|
||||
* Raw BM25 is unbounded, so each score is divided by the most the expression
|
||||
* could earn (see _getCeiling()), giving the 0-1 share of the query a document
|
||||
* carries. Both indexes are on that scale, and BM25 normalizes each by its
|
||||
* own typical document length, so a title and a 400-page PDF compare.
|
||||
*/
|
||||
Zotero.Lexical = new function () {
|
||||
// CJK scripts (Han/Hiragana/Katakana/Hangul), the same set the full-text
|
||||
|
|
@ -73,37 +59,20 @@ Zotero.Lexical = new function () {
|
|||
// word token as the index's unicode61 tokenizer produces them -- a run of
|
||||
// letters and digits, with everything else a separator. The lookahead
|
||||
// keeps CJK characters (which are also \p{L}) out of word tokens, so
|
||||
// 'covid疫情' splits into a word and a run rather than reading as one word.
|
||||
// 'covid疫情' splits into a word and a run.
|
||||
const TOKEN_RE = new RegExp(
|
||||
`(?<cjk>[${CJK_CLASS}]+)|(?<word>(?:(?![${CJK_CLASS}])[\\p{L}\\p{N}])+)`,
|
||||
'gu'
|
||||
);
|
||||
// How many times a word term is repeated in the expression, against one
|
||||
// occurrence of each adjacency phrase (see buildExpression()). BM25 sums a
|
||||
// contribution per term in the expression and counts a repeat again, which
|
||||
// is the only way to weigh one term against another in it. A phrase can
|
||||
// only match texts that match both its words, so it is always the rarer
|
||||
// term and BM25 would otherwise let it decide the ranking: unrepeated, an
|
||||
// accidental adjacency straddling two concepts ("change adaptation" in
|
||||
// "climate change adaptation in bangladesh") outweighs what the query is
|
||||
// about. Repetition trades adjacency for coverage smoothly, and this is
|
||||
// the point on that curve where a query's words still decide it.
|
||||
const ADJACENCY_REPETITION = 3;
|
||||
// What a match in each column of the item-text index is worth, in the
|
||||
// order the index declares them (see fulltext.js): a title names the work,
|
||||
// an abstract summarizes it, everything else speaks with equal voice.
|
||||
// Passed to bm25(), which spends them on how fast a match approaches the
|
||||
// ceiling rather than on the ceiling itself, so a weight can favour a
|
||||
// column without letting it score above a full match.
|
||||
// ceiling, so a weight can favour a column without letting it score
|
||||
// above a full match.
|
||||
const COLUMN_WEIGHTS = { title: 6, abstract: 4, note: 1, annotation: 2 };
|
||||
// The indexes a query is scored against: the attachment content index and
|
||||
// the item-text index, each with the CJK 2-gram table covering the same
|
||||
// documents, and the item-text tables' columns in declaration order
|
||||
// FTS5's BM25 constants, mirrored so a score built here is on the same
|
||||
// footing as one the index produced: the saturation constant K1 and the
|
||||
// (K1 + 1) factor its numerator carries, the length normalization B, and
|
||||
// the floor FTS5 puts under an inverse document frequency -- its way of
|
||||
// saying a term in more than about half a corpus separates nothing there.
|
||||
// BM25 as FTS5's bm25() computes it: the standard k1 and b, and the
|
||||
// floor it puts under a non-positive idf (a term in over half the corpus)
|
||||
const FTS5_K1 = 1.2;
|
||||
const FTS5_B = 0.75;
|
||||
const FTS5_MIN_IDF = 1e-6;
|
||||
|
|
@ -111,18 +80,25 @@ Zotero.Lexical = new function () {
|
|||
// context, narrow enough to fit on a row
|
||||
const EXCERPT_WINDOW = 200;
|
||||
|
||||
// The indexes a query is scored against. Each source is a word table and
|
||||
// a CJK 2-gram table over the same documents. `columns` names a
|
||||
// multi-column table's columns in the order the table declares them.
|
||||
const SOURCES = [
|
||||
{
|
||||
name: 'content',
|
||||
word: 'fulltextContent',
|
||||
cjk: 'fulltextContentCJK',
|
||||
columns: null
|
||||
columns: null,
|
||||
// A whole document: matches count only where they are close together
|
||||
proximity: true
|
||||
},
|
||||
{
|
||||
name: 'itemText',
|
||||
word: 'fulltextItemText',
|
||||
cjk: 'fulltextItemTextCJK',
|
||||
columns: ['title', 'abstract', 'note', 'annotation']
|
||||
columns: ['title', 'abstract', 'note', 'annotation'],
|
||||
// Short texts: a title, an abstract, an annotation
|
||||
proximity: false
|
||||
}
|
||||
];
|
||||
|
||||
|
|
@ -219,9 +195,8 @@ Zotero.Lexical = new function () {
|
|||
* Every term is joined with OR, so any of them can match and BM25 sums
|
||||
* what each contributes. The word family additionally carries a phrase
|
||||
* term for each consecutive pair of query words -- what the query asked
|
||||
* for beyond the words themselves -- and repeats each word term
|
||||
* ADJACENCY_REPETITION times to keep those phrases from deciding the
|
||||
* ranking on their own.
|
||||
* for beyond the words themselves -- with the word terms repeated to
|
||||
* keep those phrases from deciding the ranking on their own.
|
||||
*
|
||||
* Terms are all letters and digits (parseQuery tokenized them), so
|
||||
* quoting them into FTS phrases needs no escaping.
|
||||
|
|
@ -250,12 +225,17 @@ Zotero.Lexical = new function () {
|
|||
}
|
||||
}
|
||||
else {
|
||||
// Times each word is repeated against one occurrence of each phrase.
|
||||
// A repeat counts again in BM25, the only way to weigh terms in an
|
||||
// expression; unrepeated, an accidental adjacency ("change adaptation"
|
||||
// in "climate change adaptation in bangladesh") outweighs the words.
|
||||
const WORD_REPETITION = 3;
|
||||
let words = terms.filter(term => term.type == 'word');
|
||||
for (let term of terms) {
|
||||
if (term.type == 'word') {
|
||||
pieces.push({
|
||||
match: '"' + term.text + '"' + (term.prefix ? '*' : ''),
|
||||
repetitions: ADJACENCY_REPETITION
|
||||
repetitions: WORD_REPETITION
|
||||
});
|
||||
}
|
||||
else if (term.type == 'phrase') {
|
||||
|
|
@ -281,6 +261,39 @@ Zotero.Lexical = new function () {
|
|||
return { expression, pieces };
|
||||
};
|
||||
|
||||
/**
|
||||
* The FTS5 expression a document matches only where enough of a query's
|
||||
* terms occur within a window of each other: a NEAR group over the
|
||||
* required number of the family's terms, one group per way of choosing
|
||||
* them, joined with OR. Callers pass the terms BM25 scores with, so a
|
||||
* word the corpus is full of isn't required.
|
||||
*
|
||||
* @param {Object[]} terms - Terms from parseQuery()
|
||||
* @param {String} family - 'word' or 'cjk', as for buildExpression()
|
||||
* @return {String|null} - Null when the family has fewer than two terms,
|
||||
* which have nothing to be close to
|
||||
*/
|
||||
this.buildProximityExpression = function (terms, family) {
|
||||
// Tokens the terms may span: about a page
|
||||
const WINDOW = 200;
|
||||
// Share of the terms the window must hold: all of them up to three,
|
||||
// so a short query gets the words it names; a longer one may miss one
|
||||
const COVERAGE = 0.75;
|
||||
// Terms past this are ignored, bounding the number of groups
|
||||
const MAX_TERMS = 8;
|
||||
let matches = terms
|
||||
.filter(term => (term.type == 'cjk') == (family == 'cjk'))
|
||||
.slice(0, MAX_TERMS)
|
||||
.map(_termMatch);
|
||||
if (matches.length < 2) {
|
||||
return null;
|
||||
}
|
||||
let required = Math.ceil(matches.length * COVERAGE);
|
||||
return _combinations(matches, required)
|
||||
.map(group => 'NEAR(' + group.join(' ') + ', ' + WINDOW + ')')
|
||||
.join(' OR ');
|
||||
};
|
||||
|
||||
/**
|
||||
* Score a given set of items by how well their text answers a query.
|
||||
*
|
||||
|
|
@ -290,8 +303,12 @@ Zotero.Lexical = new function () {
|
|||
* as the share of the query a document carries: rarer terms move it most,
|
||||
* a document missing a term forfeits that term's share, and repetition
|
||||
* beyond the point where a text is clearly about a term adds nothing.
|
||||
* An item's score is its best across the indexes, and items below
|
||||
* SCORE_FLOOR aren't matches and aren't returned.
|
||||
* An attachment's fulltext is a whole document, and matches scattered
|
||||
* through it say nothing together, so it counts only where the query's
|
||||
* terms occur close together (see buildProximityExpression()); an
|
||||
* item's own text is short enough to count as is. An item's score is its
|
||||
* best across the indexes, and items below SCORE_FLOOR aren't matches and
|
||||
* aren't returned.
|
||||
*
|
||||
* Notes edited since their last index update, and notes not indexed yet,
|
||||
* can't be scored from the index at all; their current text is read and
|
||||
|
|
@ -303,12 +320,9 @@ Zotero.Lexical = new function () {
|
|||
* @param {Object} [options]
|
||||
* @param {Function} [options.shouldCancel] - Checked between indexes;
|
||||
* return true to abandon scoring with a ScoringCancelledError
|
||||
* @param {String[]} [options.sources] - Limit scoring to these sources by
|
||||
* name: 'content' (attachment fulltext) and/or 'itemText' (titles,
|
||||
* abstracts, notes, annotations). All sources when omitted.
|
||||
* @return {Promise<Map>} - itemID -> score (0-1, higher is better)
|
||||
*/
|
||||
this.scoreItemIDs = async function (queryText, itemIDs, { shouldCancel, sources } = {}) {
|
||||
this.scoreItemIDs = async function (queryText, itemIDs, { shouldCancel } = {}) {
|
||||
// Scores below this aren't matches and aren't returned: measured
|
||||
// against what the query could earn, this is the share of it a
|
||||
// document has to carry. Provisional until tuned against a real
|
||||
|
|
@ -327,6 +341,7 @@ Zotero.Lexical = new function () {
|
|||
throw new this.ScoringCancelledError();
|
||||
}
|
||||
};
|
||||
let scoring = await _getScoringTerms(terms);
|
||||
let candidates = new Set(itemIDs);
|
||||
let best = new Map();
|
||||
let keep = (itemID, fraction) => {
|
||||
|
|
@ -335,9 +350,6 @@ Zotero.Lexical = new function () {
|
|||
}
|
||||
};
|
||||
for (let source of SOURCES) {
|
||||
if (sources && !sources.includes(source.name)) {
|
||||
continue;
|
||||
}
|
||||
for (let family of ['word', 'cjk']) {
|
||||
checkCancel();
|
||||
let built = this.buildExpression(terms, family);
|
||||
|
|
@ -367,6 +379,19 @@ Zotero.Lexical = new function () {
|
|||
if (!rows.length) {
|
||||
continue;
|
||||
}
|
||||
let proximity = source.proximity
|
||||
&& this.buildProximityExpression(scoring, family);
|
||||
if (proximity) {
|
||||
let close = new Set(await Zotero.DB.columnQueryAsync(
|
||||
"SELECT rowid FROM ftindex." + table
|
||||
+ " WHERE " + table + " MATCH ?",
|
||||
[proximity]
|
||||
));
|
||||
rows = rows.filter(row => close.has(row.rowid));
|
||||
if (!rows.length) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let ceiling = await _getCeiling(built.pieces, table);
|
||||
// rank is negative, better more negative. The ceiling normally
|
||||
// bounds it, but a query whose every term is past FTS5's idf
|
||||
|
|
@ -406,10 +431,9 @@ Zotero.Lexical = new function () {
|
|||
* excerpts.
|
||||
*
|
||||
* Only the terms BM25 can score with are shown (see _getScoringTerms()),
|
||||
* so an excerpt points at what actually ranked the item rather than at
|
||||
* every word of the query. Each excerpt's `strength` is the share of those
|
||||
* terms it shows, so excerpts can be ordered against other relevance
|
||||
* evidence for the item.
|
||||
* so an excerpt points at what actually ranked the item. Each excerpt's
|
||||
* `strength` is the share of those terms it shows, so excerpts can be
|
||||
* ordered against other relevance evidence for the item.
|
||||
*
|
||||
* @param {String} queryText
|
||||
* @param {Number} itemID
|
||||
|
|
@ -467,30 +491,6 @@ Zotero.Lexical = new function () {
|
|||
});
|
||||
};
|
||||
|
||||
/**
|
||||
* How many terms of a query BM25 can score with (see
|
||||
* _getScoringTerms()) -- a reading of the query's size that ignores the
|
||||
* corpus-filling words the ranking ignores too
|
||||
*
|
||||
* @param {String} queryText
|
||||
* @return {Promise<Number>}
|
||||
*/
|
||||
this.getScoringTermCount = async function (queryText) {
|
||||
return (await _getScoringTerms(this.parseQuery(queryText))).length;
|
||||
};
|
||||
|
||||
/**
|
||||
* Whether every part of a query is quoted -- the query's way of asking
|
||||
* for its literal words and nothing else, wherever they appear
|
||||
*
|
||||
* @param {String} queryText
|
||||
* @return {Boolean}
|
||||
*/
|
||||
this.isFullyQuotedQuery = function (queryText) {
|
||||
let parts = Zotero.SearchConditions.parseSearchString(queryText || '');
|
||||
return parts.length > 0 && parts.every(part => part.inQuotes);
|
||||
};
|
||||
|
||||
/**
|
||||
* How much of a query each of the given texts carries, on the same 0-1
|
||||
* scale as scoreItemIDs(): each text's BM25 score over the terms BM25 can
|
||||
|
|
@ -504,8 +504,7 @@ Zotero.Lexical = new function () {
|
|||
* document is about appears in every passage, so it says nothing about
|
||||
* which passage answers the query -- however rare it is in the corpus
|
||||
* that surfaced the document. Length is normalized against the texts'
|
||||
* own average, so a passage is long or short relative to its siblings
|
||||
* rather than to whole documents.
|
||||
* own average, so a passage is long or short relative to its siblings.
|
||||
*
|
||||
* @param {String} queryText
|
||||
* @param {String[]} texts
|
||||
|
|
@ -535,7 +534,7 @@ Zotero.Lexical = new function () {
|
|||
});
|
||||
// The smoothed BM25 idf (the +1 keeps it positive), so a term in
|
||||
// every text still separates texts of one and of many occurrences a
|
||||
// little, rather than flipping negative
|
||||
// little
|
||||
let weighted = terms.map((term) => {
|
||||
let df = counts.reduce((sum, perText) => sum + (perText.has(term) ? 1 : 0), 0);
|
||||
return {
|
||||
|
|
@ -603,6 +602,20 @@ Zotero.Lexical = new function () {
|
|||
return best.bounds;
|
||||
};
|
||||
|
||||
// Every way of choosing `size` of the items, in order
|
||||
function _combinations(items, size) {
|
||||
if (size == 0) {
|
||||
return [[]];
|
||||
}
|
||||
let groups = [];
|
||||
items.forEach((item, i) => {
|
||||
for (let rest of _combinations(items.slice(i + 1), size - 1)) {
|
||||
groups.push([item, ...rest]);
|
||||
}
|
||||
});
|
||||
return groups;
|
||||
}
|
||||
|
||||
// The FTS5 MATCH expression matching one term, as buildExpression() writes
|
||||
// it, for counting the documents a term is in
|
||||
function _termMatch(term) {
|
||||
|
|
@ -878,7 +891,7 @@ Zotero.Lexical = new function () {
|
|||
m => m.start < best.bounds.start || m.end > best.bounds.end
|
||||
);
|
||||
// A match wider than the window sits outside every window's
|
||||
// bounds; nothing can show it, so drop it rather than loop on it
|
||||
// bounds; nothing can show it, so drop it
|
||||
if (left.length == remaining.length) {
|
||||
left = remaining.slice(1);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ describe("Zotero.BestMatch", function () {
|
|||
(sum, [fraction, rank]) => sum + fraction / (60 + rank), 0) / (2 / 61);
|
||||
}
|
||||
|
||||
function stubEngines({ enabled, lexical, semantic, fraction, termCount, quoted }) {
|
||||
function stubEngines({ enabled, lexical, semantic, fraction }) {
|
||||
stubs.push(sinon.stub(Zotero.Embeddings, 'isEnabled').returns(enabled));
|
||||
// Semantic scores pass through the display band unchanged unless a
|
||||
// test maps them, so expected fused values compute from the stubbed
|
||||
|
|
@ -21,12 +21,6 @@ describe("Zotero.BestMatch", function () {
|
|||
if (lexical) {
|
||||
stubs.push(sinon.stub(Zotero.Lexical, 'scoreItemIDs').callsFake(lexical));
|
||||
}
|
||||
// A keyword query unless the test says otherwise, so the lexical
|
||||
// engine searches everything
|
||||
stubs.push(sinon.stub(Zotero.Lexical, 'getScoringTermCount')
|
||||
.resolves(termCount ?? 1));
|
||||
stubs.push(sinon.stub(Zotero.Lexical, 'isFullyQuotedQuery')
|
||||
.returns(quoted ?? false));
|
||||
// Per-test semantic fakes return bare score Maps; wrap them in the
|
||||
// engine's { scores, previewableIDs } envelope
|
||||
if (semantic) {
|
||||
|
|
@ -74,61 +68,6 @@ describe("Zotero.BestMatch", function () {
|
|||
assert.isAbove(scores.get(1), scores.get(3));
|
||||
});
|
||||
|
||||
it("should search only the items' own text for a longer query", async function () {
|
||||
let lexicalStub = sinon.stub(Zotero.Lexical, 'scoreItemIDs').resolves(new Map());
|
||||
stubs.push(lexicalStub);
|
||||
stubEngines({
|
||||
enabled: true,
|
||||
semantic: async () => new Map(),
|
||||
termCount: 3
|
||||
});
|
||||
|
||||
await Zotero.BestMatch.scoreItemIDs('owl migration patterns', [1]);
|
||||
assert.deepEqual(lexicalStub.firstCall.args[2].sources, ['itemText']);
|
||||
|
||||
// A keyword query searches everything
|
||||
lexicalStub.resetHistory();
|
||||
Zotero.Lexical.getScoringTermCount.resolves(2);
|
||||
await Zotero.BestMatch.scoreItemIDs('owl migration', [1]);
|
||||
assert.isUndefined(lexicalStub.firstCall.args[2].sources);
|
||||
});
|
||||
|
||||
it("should search everything for a fully quoted query", async function () {
|
||||
let lexicalStub = sinon.stub(Zotero.Lexical, 'scoreItemIDs').resolves(new Map());
|
||||
stubs.push(lexicalStub);
|
||||
stubEngines({
|
||||
enabled: true,
|
||||
semantic: async () => new Map(),
|
||||
termCount: 3,
|
||||
quoted: true
|
||||
});
|
||||
|
||||
await Zotero.BestMatch.scoreItemIDs('"owl" "migration" "patterns"', [1]);
|
||||
assert.isUndefined(lexicalStub.firstCall.args[2].sources);
|
||||
});
|
||||
|
||||
it("should fall back to lexical over everything when the index isn't ready", async function () {
|
||||
stubEngines({
|
||||
enabled: true,
|
||||
// The items' own text holds one match; the fulltext another
|
||||
lexical: async (queryText, itemIDs, options) => (
|
||||
options.sources
|
||||
? new Map([[1, 0.5]])
|
||||
: new Map([[1, 0.5], [2, 0.4]])
|
||||
),
|
||||
semantic: async () => {
|
||||
throw new Zotero.Embeddings.IndexNotReadyError('test');
|
||||
},
|
||||
termCount: 3
|
||||
});
|
||||
|
||||
let { scores, matches } = await Zotero.BestMatch.scoreItemIDs(
|
||||
'owl migration patterns', [1, 2]);
|
||||
// With no model to read the fulltext, lexical covers all of it
|
||||
assert.deepEqual([...scores.entries()], [[1, 0.5], [2, 0.4]]);
|
||||
assert.sameMembers([...matches.lexical], [1, 2]);
|
||||
});
|
||||
|
||||
it("should keep the model's ordering above the display band", async function () {
|
||||
stubEngines({
|
||||
enabled: true,
|
||||
|
|
@ -358,8 +297,6 @@ describe("Zotero.BestMatch", function () {
|
|||
stubs.push(sinon.stub(Zotero.Embeddings, 'isEnabled').returns(true));
|
||||
stubs.push(sinon.stub(Zotero.Embeddings, 'getScoreFraction')
|
||||
.callsFake(score => score));
|
||||
stubs.push(sinon.stub(Zotero.Lexical, 'getScoringTermCount').resolves(1));
|
||||
stubs.push(sinon.stub(Zotero.Lexical, 'isFullyQuotedQuery').returns(false));
|
||||
stubs.push(sinon.stub(Zotero.Lexical, 'scoreItemIDs')
|
||||
.resolves(new Map([[1, 0.8], [2, 0.9]])));
|
||||
stubs.push(sinon.stub(Zotero.Embeddings, 'scoreItemIDs').resolves({
|
||||
|
|
|
|||
|
|
@ -153,6 +153,45 @@ describe("Zotero.Lexical", function () {
|
|||
});
|
||||
});
|
||||
|
||||
describe("#buildProximityExpression()", function () {
|
||||
function expression(queryText, family = 'word') {
|
||||
return Zotero.Lexical.buildProximityExpression(
|
||||
Zotero.Lexical.parseQuery(queryText), family);
|
||||
}
|
||||
|
||||
it("should gather every word of a short query in one window", function () {
|
||||
assert.equal(expression("fall communism "),
|
||||
'NEAR("fall" "communism", 200)');
|
||||
assert.equal(expression("owl migration patterns "),
|
||||
'NEAR("owl" "migration" "patterns", 200)');
|
||||
});
|
||||
|
||||
it("should let a longer query miss a word, in any position", function () {
|
||||
assert.equal(expression("a1 b2 c3 d4 "), [
|
||||
'NEAR("a1" "b2" "c3", 200)',
|
||||
'NEAR("a1" "b2" "d4", 200)',
|
||||
'NEAR("a1" "c3" "d4", 200)',
|
||||
'NEAR("b2" "c3" "d4", 200)'
|
||||
].join(' OR '));
|
||||
});
|
||||
|
||||
it("should carry prefixes and phrases as terms", function () {
|
||||
assert.equal(expression('owl* "barn owl" '),
|
||||
'NEAR("owl"* "barn owl", 200)');
|
||||
});
|
||||
|
||||
it("should gate each family on its own terms", function () {
|
||||
assert.equal(expression("owl 猫头鹰 ", 'word'), null);
|
||||
assert.equal(expression("猫头鹰 迁徙 ", 'cjk'),
|
||||
'NEAR("猫头 头鹰" "迁徙", 200)');
|
||||
});
|
||||
|
||||
it("should return null for a single term, which gathers anywhere", function () {
|
||||
assert.isNull(expression("owl "));
|
||||
assert.isNull(expression(""));
|
||||
});
|
||||
});
|
||||
|
||||
describe("#scoreItemIDs()", function () {
|
||||
const BASE_ROWID = 940000000;
|
||||
var inserted = [];
|
||||
|
|
@ -171,6 +210,19 @@ describe("Zotero.Lexical", function () {
|
|||
return rowid;
|
||||
}
|
||||
|
||||
// A corpus of unrelated documents and items, so a test's words are
|
||||
// rare in it: in a corpus of two, every query word is in more than
|
||||
// half of it, and FTS5 weighs such a word as nothing
|
||||
before(async function () {
|
||||
let subjects = ['tides', 'granite', 'sonnets', 'bridges', 'yeast',
|
||||
'glaciers', 'violins', 'ledgers', 'orchards', 'compasses'];
|
||||
for (let i = 0; i < subjects.length; i++) {
|
||||
await addContentDoc(900 + i,
|
||||
`notes on ${subjects[i]} and their study through the years`);
|
||||
await createDataObject('item', { title: `A survey of ${subjects[i]}` });
|
||||
}
|
||||
});
|
||||
|
||||
after(async function () {
|
||||
for (let rowid of inserted) {
|
||||
await Zotero.DB.queryAsync(
|
||||
|
|
@ -180,34 +232,38 @@ describe("Zotero.Lexical", function () {
|
|||
}
|
||||
});
|
||||
|
||||
it("should limit scoring to the given sources", async function () {
|
||||
let doc = await addContentDoc(46, 'lexsource in fulltext alone');
|
||||
let item = await createDataObject('item', { title: 'Lexsource in a title' });
|
||||
it("should match a document only where the query's words share a passage", async function () {
|
||||
let near = await addContentDoc(60,
|
||||
'the lexproxfall of lexproxcommunism in eastern europe');
|
||||
let far = await addContentDoc(61,
|
||||
'lexproxcommunism spread widely ' + 'filler '.repeat(300)
|
||||
+ 'in the lexproxfall season the leaves drop');
|
||||
// An item's own text is a passage already, so one word matches it
|
||||
let item = await createDataObject('item',
|
||||
{ title: 'Lexproxcommunism in the twentieth century' });
|
||||
|
||||
let all = await Zotero.Lexical.scoreItemIDs('lexsource', [doc, item.id]);
|
||||
assert.isTrue(all.has(doc));
|
||||
assert.isTrue(all.has(item.id));
|
||||
|
||||
// The items' own text doesn't include attachment fulltext
|
||||
let own = await Zotero.Lexical.scoreItemIDs('lexsource', [doc, item.id],
|
||||
{ sources: ['itemText'] });
|
||||
assert.isFalse(own.has(doc));
|
||||
assert.isTrue(own.has(item.id));
|
||||
let scores = await Zotero.Lexical.scoreItemIDs(
|
||||
'lexproxfall lexproxcommunism', [near, far, item.id]);
|
||||
assert.isTrue(scores.has(near));
|
||||
assert.isFalse(scores.has(far));
|
||||
assert.isTrue(scores.has(item.id));
|
||||
});
|
||||
|
||||
it("should limit scoring to the given sources", async function () {
|
||||
let doc = await addContentDoc(46, 'lexsource in fulltext alone');
|
||||
let item = await createDataObject('item', { title: 'Lexsource in a title' });
|
||||
it("should admit a document saying most of a longer query in one passage", async function () {
|
||||
let most = await addContentDoc(62,
|
||||
'lexpcova lexpcovb and lexpcovc discussed here');
|
||||
let half = await addContentDoc(63,
|
||||
'lexpcova lexpcovb discussed here');
|
||||
let scattered = await addContentDoc(64,
|
||||
'lexpcova lexpcovb ' + 'filler '.repeat(300) + 'lexpcovc lexpcovd');
|
||||
|
||||
let all = await Zotero.Lexical.scoreItemIDs('lexsource', [doc, item.id]);
|
||||
assert.isTrue(all.has(doc));
|
||||
assert.isTrue(all.has(item.id));
|
||||
|
||||
// The items' own text doesn't include attachment fulltext
|
||||
let own = await Zotero.Lexical.scoreItemIDs('lexsource', [doc, item.id],
|
||||
{ sources: ['itemText'] });
|
||||
assert.isFalse(own.has(doc));
|
||||
assert.isTrue(own.has(item.id));
|
||||
let scores = await Zotero.Lexical.scoreItemIDs(
|
||||
'lexpcova lexpcovb lexpcovc lexpcovd', [most, half, scattered]);
|
||||
// Three of four words in one passage is saying the query...
|
||||
assert.isTrue(scores.has(most));
|
||||
// ...two of four isn't, and neither is all four spread apart
|
||||
assert.isFalse(scores.has(half));
|
||||
assert.isFalse(scores.has(scattered));
|
||||
});
|
||||
|
||||
it("should return scores as 0-1 fractions", async function () {
|
||||
|
|
@ -336,24 +392,6 @@ describe("Zotero.Lexical", function () {
|
|||
});
|
||||
});
|
||||
|
||||
describe("#getScoringTermCount()", function () {
|
||||
it("should count the terms BM25 can score with", async function () {
|
||||
assert.equal(
|
||||
await Zotero.Lexical.getScoringTermCount('lexcountaaa lexcountbbb'), 2);
|
||||
assert.equal(await Zotero.Lexical.getScoringTermCount(''), 0);
|
||||
});
|
||||
});
|
||||
|
||||
describe("#isFullyQuotedQuery()", function () {
|
||||
it("should be true only when every part is quoted", function () {
|
||||
assert.isTrue(Zotero.Lexical.isFullyQuotedQuery('"owl migration"'));
|
||||
assert.isTrue(Zotero.Lexical.isFullyQuotedQuery('"owl" "migration"'));
|
||||
assert.isFalse(Zotero.Lexical.isFullyQuotedQuery('"owl" migration'));
|
||||
assert.isFalse(Zotero.Lexical.isFullyQuotedQuery('owl migration'));
|
||||
assert.isFalse(Zotero.Lexical.isFullyQuotedQuery(''));
|
||||
});
|
||||
});
|
||||
|
||||
describe("#getMatchingExcerpts()", function () {
|
||||
function rangeTexts(excerpt) {
|
||||
return excerpt.ranges.map(([start, end]) => excerpt.text.slice(start, end));
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue