mirror of
https://github.com/zotero/zotero.git
synced 2026-10-03 02:21:49 +00:00
Duplicates: allow to search for duplicates by specific item
This commit is contained in:
parent
5977f1c917
commit
80b82fc953
2 changed files with 531 additions and 264 deletions
|
|
@ -67,6 +67,34 @@ Zotero.Duplicates.prototype._getLibraryCondition = function (field = 'libraryID'
|
|||
};
|
||||
};
|
||||
|
||||
Zotero.Duplicates.normalizeString = function (str) {
|
||||
// Make sure we have a string and not an integer
|
||||
str = str + "";
|
||||
|
||||
if (str === "") {
|
||||
return "";
|
||||
}
|
||||
|
||||
str = Zotero.Utilities.removeDiacritics(str)
|
||||
.replace(/[ !-/:-@[-`{-~]+/g, ' ') // Convert (ASCII) punctuation to spaces
|
||||
.trim()
|
||||
.toLowerCase();
|
||||
|
||||
return str;
|
||||
};
|
||||
|
||||
Zotero.Duplicates._sortByValue = function (a, b) {
|
||||
if ((a.value === null && b.value !== null)
|
||||
|| (a.value === undefined && b.value !== undefined)
|
||||
|| a.value < b.value) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
if (a.value === b.value) return 0;
|
||||
|
||||
return 1;
|
||||
};
|
||||
|
||||
/**
|
||||
* Get duplicates, populate a temporary table, and return a search based
|
||||
* on that table
|
||||
|
|
@ -128,101 +156,144 @@ Zotero.Duplicates.prototype._getObjectFromID = function (id) {
|
|||
}
|
||||
|
||||
|
||||
Zotero.Duplicates.prototype._findDuplicates = async function () {
|
||||
Zotero.debug("Finding duplicates");
|
||||
/**
|
||||
* The comparison function for title-based duplicate matching.
|
||||
*
|
||||
* Reads metadata directly from the row objects (which are enriched with
|
||||
* doi/isbn/year/creators in _loadCaches and findDuplicatesOf), so it has
|
||||
* no dependency on instance caches.
|
||||
*
|
||||
* Assumes rows are sorted by normalized title. Returns:
|
||||
* -1: not a match, stop comparing (title mismatch in sorted order)
|
||||
* 0: not a match, but keep looking (title matches but metadata conflicts)
|
||||
* 1: match
|
||||
*
|
||||
* @param {Object} a - Enriched row {itemID, value, doi?, isbn?, year?, creators?}
|
||||
* @param {Object} b - Enriched row {itemID, value, doi?, isbn?, year?, creators?}
|
||||
* @return {Integer}
|
||||
*/
|
||||
Zotero.Duplicates._compareRows = function (a, b) {
|
||||
var aTitle = a.value;
|
||||
var bTitle = b.value;
|
||||
|
||||
var start = Date.now();
|
||||
|
||||
var self = this;
|
||||
|
||||
this._sets = new Zotero.DisjointSetForest;
|
||||
var sets = this._sets;
|
||||
|
||||
function normalizeString(str) {
|
||||
// Make sure we have a string and not an integer
|
||||
str = str + "";
|
||||
|
||||
if (str === "") {
|
||||
return "";
|
||||
}
|
||||
|
||||
str = Zotero.Utilities.removeDiacritics(str)
|
||||
.replace(/[ !-/:-@[-`{-~]+/g, ' ') // Convert (ASCII) punctuation to spaces
|
||||
.trim()
|
||||
.toLowerCase();
|
||||
|
||||
return str;
|
||||
// If we stripped one of the strings completely, we can't compare them
|
||||
if (!aTitle || !bTitle) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
function sortByValue(a, b) {
|
||||
if((a.value === null && b.value !== null)
|
||||
|| (a.value === undefined && b.value !== undefined)
|
||||
|| a.value < b.value) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
if(a.value === b.value) return 0;
|
||||
|
||||
if (aTitle !== bTitle) {
|
||||
return -1; // everything is sorted by title, so if this mismatches, everything following will too
|
||||
}
|
||||
|
||||
// If both items have a DOI and they don't match, it's not a dupe
|
||||
if (a.doi && b.doi && a.doi != b.doi) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// If both items have an ISBN and they don't match, it's not a dupe
|
||||
if (a.isbn && b.isbn && a.isbn != b.isbn) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// If both items have a year and they're off by more than one, it's not a dupe
|
||||
if (a.year && b.year && Math.abs(a.year - b.year) > 1) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Match if neither has creators
|
||||
if (!a.creators && !b.creators) {
|
||||
return 1;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {Function} compareRows Comparison function, if not exact match
|
||||
* @param {Boolean} reprocessMatches Compare every row against every other,
|
||||
* without skipping ahead to the last match.
|
||||
* This is necessary for multi-dimensional
|
||||
* matches such as title + at least one creator.
|
||||
* Without it, only one set of matches would be
|
||||
* found per matching title, since items with
|
||||
* different creators wouldn't match the first
|
||||
* set and the next start row would be a
|
||||
* different title.
|
||||
*/
|
||||
function processRows(rows, compareRows, reprocessMatches) {
|
||||
if (!rows.length) {
|
||||
return;
|
||||
}
|
||||
// One has creators and the other doesn't — not a dupe
|
||||
if (!a.creators || !b.creators) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Check for at least one match on last name + first initial of first name
|
||||
for (let i = 0; i < a.creators.length; i++) {
|
||||
let aCreator = a.creators[i];
|
||||
let aLastName = aCreator.lastName;
|
||||
let aFirstInitial = aCreator.firstInitial || "";
|
||||
|
||||
for (var i = 0, len = rows.length; i < len; i++) {
|
||||
var j = i + 1, lastMatch = false;
|
||||
while (j < len) {
|
||||
if (compareRows) {
|
||||
var match = compareRows(rows[i], rows[j]);
|
||||
// Not a match, and don't try any more with this i value
|
||||
if (match == -1) {
|
||||
break;
|
||||
}
|
||||
// Not a match, but keep looking
|
||||
if (match == 0) {
|
||||
j++;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
// If no comparison function, check for exact match
|
||||
else {
|
||||
if (!rows[i].value || !rows[j].value
|
||||
|| (rows[i].value !== rows[j].value)
|
||||
) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
sets.union(
|
||||
self._getObjectFromID(rows[i].itemID),
|
||||
self._getObjectFromID(rows[j].itemID)
|
||||
);
|
||||
|
||||
lastMatch = j;
|
||||
j++;
|
||||
}
|
||||
if (!reprocessMatches && lastMatch) {
|
||||
i = lastMatch;
|
||||
for (let j = 0; j < b.creators.length; j++) {
|
||||
let bCreator = b.creators[j];
|
||||
let bLastName = bCreator.lastName;
|
||||
let bFirstInitial = bCreator.firstInitial || "";
|
||||
|
||||
if (aLastName === bLastName && aFirstInitial === bFirstInitial) {
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* Check if a target row has duplicates among the given rows.
|
||||
*
|
||||
* This is the inner loop of processRows, extracted so it can be reused
|
||||
* by findDuplicatesOf. Rows must be sorted by value.
|
||||
*
|
||||
* @param {Object} targetRow - Row with .itemID and .value
|
||||
* @param {Object[]} rows - Sorted rows to compare against
|
||||
* @param {Function} [compareRows] - Comparison function returning -1/0/1.
|
||||
* If omitted, checks for exact value match.
|
||||
* @return {Object[]} - Array of matching rows
|
||||
*/
|
||||
Zotero.Duplicates._checkIfDuplicate = function (targetRow, rows, compareRows) {
|
||||
let matches = [];
|
||||
for (let j = 0; j < rows.length; j++) {
|
||||
if (compareRows) {
|
||||
let match = compareRows(targetRow, rows[j]);
|
||||
// Not a match, and don't try any more
|
||||
if (match == -1) {
|
||||
break;
|
||||
}
|
||||
// Not a match, but keep looking
|
||||
if (match == 0) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
// If no comparison function, check for exact match
|
||||
else {
|
||||
if (!targetRow.value || !rows[j].value
|
||||
|| (targetRow.value !== rows[j].value)
|
||||
) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
matches.push(rows[j]);
|
||||
}
|
||||
return matches;
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* Load all data needed for duplicate detection from the database.
|
||||
*
|
||||
* Populates:
|
||||
* this._isbnRows - sorted [{itemID, value}] for ISBN exact-match pass
|
||||
* this._doiRows - sorted [{itemID, value}] for DOI exact-match pass
|
||||
* this._titleRows - sorted enriched rows for title+creators pass:
|
||||
* [{itemID, value, doi?, isbn?, year?, creators?}]
|
||||
* this._itemCache - {itemID: {doi?, isbn?, year?, creators?}} — consolidated
|
||||
* metadata used to enrich title rows
|
||||
*/
|
||||
Zotero.Duplicates.prototype._loadCaches = async function () {
|
||||
var normalizeString = Zotero.Duplicates.normalizeString;
|
||||
var sortByValue = Zotero.Duplicates._sortByValue;
|
||||
|
||||
let libraryCondition = this._getLibraryCondition();
|
||||
|
||||
this._itemCache = {};
|
||||
var getCacheEntry = (itemID) => {
|
||||
if (!this._itemCache[itemID]) this._itemCache[itemID] = {};
|
||||
return this._itemCache[itemID];
|
||||
};
|
||||
|
||||
// Match books by ISBN
|
||||
var sql = "SELECT itemID, value FROM items JOIN itemData USING (itemID) "
|
||||
+ "JOIN itemDataValues USING (valueID) "
|
||||
|
|
@ -236,31 +307,23 @@ Zotero.Duplicates.prototype._findDuplicates = async function () {
|
|||
Zotero.ItemFields.getID('ISBN')
|
||||
]
|
||||
);
|
||||
var isbnCache = {};
|
||||
if (rows.length) {
|
||||
let newRows = [];
|
||||
for (let i = 0; i < rows.length; i++) {
|
||||
let row = rows[i];
|
||||
let newVal = Zotero.Utilities.cleanISBN('' + row.value);
|
||||
if (!newVal) continue;
|
||||
// Canonicalize to ISBN-13 so an ISBN-10 and its ISBN-13 equivalent match
|
||||
newVal = Zotero.Utilities.toISBN13(newVal);
|
||||
isbnCache[row.itemID] = newVal;
|
||||
newRows.push({
|
||||
itemID: row.itemID,
|
||||
value: newVal
|
||||
});
|
||||
}
|
||||
newRows.sort(sortByValue);
|
||||
processRows(newRows);
|
||||
this._isbnRows = [];
|
||||
for (let row of rows) {
|
||||
let cleaned = Zotero.Utilities.cleanISBN('' + row.value);
|
||||
if (!cleaned) continue;
|
||||
// Canonicalize to ISBN-13 so an ISBN-10 and its ISBN-13 equivalent match
|
||||
cleaned = Zotero.Utilities.toISBN13(cleaned);
|
||||
getCacheEntry(row.itemID).isbn = cleaned;
|
||||
this._isbnRows.push({ itemID: row.itemID, value: cleaned });
|
||||
}
|
||||
this._isbnRows.sort(sortByValue);
|
||||
|
||||
// DOI
|
||||
var sql = "SELECT itemID, value FROM items JOIN itemData USING (itemID) "
|
||||
+ "JOIN itemDataValues USING (valueID) "
|
||||
+ `WHERE ${libraryCondition.sql} AND fieldID=? AND value LIKE ? `
|
||||
+ "AND itemID NOT IN (SELECT itemID FROM deletedItems)";
|
||||
var rows = await Zotero.DB.queryAsync(
|
||||
sql = "SELECT itemID, value FROM items JOIN itemData USING (itemID) "
|
||||
+ "JOIN itemDataValues USING (valueID) "
|
||||
+ `WHERE ${libraryCondition.sql} AND fieldID=? AND value LIKE ? `
|
||||
+ "AND itemID NOT IN (SELECT itemID FROM deletedItems)";
|
||||
rows = await Zotero.DB.queryAsync(
|
||||
sql,
|
||||
[
|
||||
...libraryCondition.params,
|
||||
|
|
@ -268,190 +331,242 @@ Zotero.Duplicates.prototype._findDuplicates = async function () {
|
|||
'10.%'
|
||||
]
|
||||
);
|
||||
var doiCache = {};
|
||||
if (rows.length) {
|
||||
let newRows = [];
|
||||
for (let i = 0; i < rows.length; i++) {
|
||||
let row = rows[i];
|
||||
// DOIs are case insensitive
|
||||
let newVal = (row.value + '').trim().toUpperCase();
|
||||
doiCache[row.itemID] = newVal;
|
||||
newRows.push({
|
||||
itemID: row.itemID,
|
||||
value: newVal
|
||||
});
|
||||
}
|
||||
newRows.sort(sortByValue);
|
||||
processRows(newRows);
|
||||
this._doiRows = [];
|
||||
for (let row of rows) {
|
||||
// DOIs are case insensitive
|
||||
let doi = (row.value + '').trim().toUpperCase();
|
||||
getCacheEntry(row.itemID).doi = doi;
|
||||
this._doiRows.push({ itemID: row.itemID, value: doi });
|
||||
}
|
||||
this._doiRows.sort(sortByValue);
|
||||
|
||||
// Get years
|
||||
var dateFields = [
|
||||
Zotero.ItemFields.getID('date'),
|
||||
...Zotero.ItemFields.getTypeFieldsFromBase('date')
|
||||
];
|
||||
var sql = "SELECT itemID, SUBSTR(value, 1, 4) AS year FROM items "
|
||||
+ "JOIN itemData USING (itemID) "
|
||||
+ "JOIN itemDataValues USING (valueID) "
|
||||
+ `WHERE ${libraryCondition.sql} AND fieldID IN (`
|
||||
+ dateFields.map(() => '?').join() + ") "
|
||||
+ "AND SUBSTR(value, 1, 4) != '0000' "
|
||||
+ "AND itemID NOT IN (SELECT itemID FROM deletedItems) "
|
||||
+ "ORDER BY value";
|
||||
var rows = await Zotero.DB.queryAsync(sql, [...libraryCondition.params, ...dateFields]);
|
||||
var yearCache = {};
|
||||
for (let i = 0; i < rows.length; i++) {
|
||||
let row = rows[i];
|
||||
yearCache[row.itemID] = row.year;
|
||||
sql = "SELECT itemID, SUBSTR(value, 1, 4) AS year FROM items "
|
||||
+ "JOIN itemData USING (itemID) "
|
||||
+ "JOIN itemDataValues USING (valueID) "
|
||||
+ `WHERE ${libraryCondition.sql} AND fieldID IN (`
|
||||
+ dateFields.map(() => '?').join() + ") "
|
||||
+ "AND SUBSTR(value, 1, 4) != '0000' "
|
||||
+ "AND itemID NOT IN (SELECT itemID FROM deletedItems) "
|
||||
+ "ORDER BY value";
|
||||
rows = await Zotero.DB.queryAsync(sql, [...libraryCondition.params, ...dateFields]);
|
||||
for (let row of rows) {
|
||||
getCacheEntry(row.itemID).year = row.year;
|
||||
}
|
||||
|
||||
var itemTypeAttachment = Zotero.ItemTypes.getID('attachment');
|
||||
var itemTypeNote = Zotero.ItemTypes.getID('note');
|
||||
|
||||
// Get all creators and group by itemID
|
||||
sql = "SELECT itemID, lastName, firstName, fieldMode FROM items "
|
||||
+ "JOIN itemCreators USING (itemID) "
|
||||
+ "JOIN creators USING (creatorID) "
|
||||
+ `WHERE ${libraryCondition.sql} AND itemTypeID NOT IN (${itemTypeAttachment}, ${itemTypeNote}) AND `
|
||||
+ "itemID NOT IN (SELECT itemID FROM deletedItems)"
|
||||
+ "ORDER BY itemID, orderIndex";
|
||||
let creatorRows = await Zotero.DB.queryAsync(sql, libraryCondition.params);
|
||||
for (let row of creatorRows) {
|
||||
let entry = getCacheEntry(row.itemID);
|
||||
if (!entry.creators) entry.creators = [];
|
||||
entry.creators.push({
|
||||
lastName: normalizeString(row.lastName),
|
||||
firstInitial: row.fieldMode == 0 ? normalizeString(row.firstName).charAt(0) : false
|
||||
});
|
||||
}
|
||||
|
||||
// Match on normalized title
|
||||
var titleIDs = Zotero.ItemFields.getTypeFieldsFromBase('title');
|
||||
titleIDs.push(Zotero.ItemFields.getID('title'));
|
||||
var sql = "SELECT itemID, value FROM items JOIN itemData USING (itemID) "
|
||||
+ "JOIN itemDataValues USING (valueID) "
|
||||
+ `WHERE ${libraryCondition.sql} AND fieldID IN `
|
||||
+ "(" + titleIDs.join(', ') + ") "
|
||||
+ `AND itemTypeID NOT IN (${itemTypeAttachment}, ${itemTypeNote}) `
|
||||
+ "AND itemID NOT IN (SELECT itemID FROM deletedItems)";
|
||||
var rows = await Zotero.DB.queryAsync(sql, libraryCondition.params);
|
||||
if (rows.length) {
|
||||
//normalize all values ahead of time
|
||||
rows = rows.map(function (row) {
|
||||
return {
|
||||
itemID: row.itemID,
|
||||
value: normalizeString(row.value)
|
||||
};
|
||||
});
|
||||
//sort rows by normalized values
|
||||
rows.sort(sortByValue);
|
||||
|
||||
// Get all creators and separate by itemID
|
||||
//
|
||||
// We won't need all of these, but otherwise we would have to make processRows()
|
||||
// asynchronous, which would be too slow
|
||||
let creatorRowsCache = {};
|
||||
let sql = "SELECT itemID, lastName, firstName, fieldMode FROM items "
|
||||
+ "JOIN itemCreators USING (itemID) "
|
||||
+ "JOIN creators USING (creatorID) "
|
||||
+ `WHERE ${libraryCondition.sql} AND itemTypeID NOT IN (${itemTypeAttachment}, ${itemTypeNote}) AND `
|
||||
+ "itemID NOT IN (SELECT itemID FROM deletedItems)"
|
||||
+ "ORDER BY itemID, orderIndex";
|
||||
let creatorRows = await Zotero.DB.queryAsync(sql, libraryCondition.params);
|
||||
let lastItemID;
|
||||
let itemCreators = [];
|
||||
for (let i = 0; i < creatorRows.length; i++) {
|
||||
let row = creatorRows[i];
|
||||
if (lastItemID && row.itemID != lastItemID) {
|
||||
if (itemCreators.length) {
|
||||
creatorRowsCache[lastItemID] = itemCreators;
|
||||
itemCreators = [];
|
||||
}
|
||||
}
|
||||
|
||||
lastItemID = row.itemID;
|
||||
|
||||
itemCreators.push({
|
||||
lastName: normalizeString(row.lastName),
|
||||
firstInitial: row.fieldMode == 0 ? normalizeString(row.firstName).charAt(0) : false
|
||||
});
|
||||
sql = "SELECT itemID, value FROM items JOIN itemData USING (itemID) "
|
||||
+ "JOIN itemDataValues USING (valueID) "
|
||||
+ `WHERE ${libraryCondition.sql} AND fieldID IN `
|
||||
+ "(" + titleIDs.join(', ') + ") "
|
||||
+ `AND itemTypeID NOT IN (${itemTypeAttachment}, ${itemTypeNote}) `
|
||||
+ "AND itemID NOT IN (SELECT itemID FROM deletedItems)";
|
||||
rows = await Zotero.DB.queryAsync(sql, libraryCondition.params);
|
||||
// Normalize titles and enrich with metadata from the cache
|
||||
this._titleRows = rows.map((row) => {
|
||||
let entry = this._itemCache[row.itemID] || {};
|
||||
return {
|
||||
itemID: row.itemID,
|
||||
value: normalizeString(row.value),
|
||||
...entry
|
||||
};
|
||||
});
|
||||
// Sort rows by normalized values
|
||||
this._titleRows.sort(sortByValue);
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* Process sorted rows, finding duplicates and unioning them into sets.
|
||||
*
|
||||
* @param {Object[]} rows - Sorted rows with .itemID and .value
|
||||
* @param {Function} [compareRows] - Comparison function returning -1/0/1.
|
||||
* If omitted, checks for exact value match.
|
||||
* @param {Boolean} [reprocessMatches] - If true, don't skip ahead past matches.
|
||||
* Needed for multi-dimensional comparisons
|
||||
* (e.g. title + creators) where items with
|
||||
* the same title but different creators
|
||||
* must still be compared individually.
|
||||
*/
|
||||
Zotero.Duplicates.prototype._processRows = function (rows, compareRows, reprocessMatches) {
|
||||
for (let i = 0, len = rows.length; i < len; i++) {
|
||||
let matches = Zotero.Duplicates._checkIfDuplicate(
|
||||
rows[i], rows.slice(i + 1), compareRows
|
||||
);
|
||||
for (let m of matches) {
|
||||
this._sets.union(
|
||||
this._getObjectFromID(rows[i].itemID),
|
||||
this._getObjectFromID(m.itemID)
|
||||
);
|
||||
}
|
||||
// Add final item creators
|
||||
if (itemCreators.length) {
|
||||
creatorRowsCache[lastItemID] = itemCreators;
|
||||
if (!reprocessMatches && matches.length) {
|
||||
i += matches.length;
|
||||
}
|
||||
|
||||
processRows(rows, function (a, b) {
|
||||
var aTitle = a.value;
|
||||
var bTitle = b.value;
|
||||
|
||||
// If we stripped one of the strings completely, we can't compare them
|
||||
if(!aTitle || !bTitle) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
if (aTitle !== bTitle) {
|
||||
return -1; //everything is sorted by title, so if this mismatches, everything following will too
|
||||
}
|
||||
|
||||
// If both items have a DOI and they don't match, it's not a dupe
|
||||
if (typeof doiCache[a.itemID] != 'undefined'
|
||||
&& typeof doiCache[b.itemID] != 'undefined'
|
||||
&& doiCache[a.itemID] != doiCache[b.itemID]) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// If both items have an ISBN and they don't match, it's not a dupe
|
||||
if (typeof isbnCache[a.itemID] != 'undefined'
|
||||
&& typeof isbnCache[b.itemID] != 'undefined'
|
||||
&& isbnCache[a.itemID] != isbnCache[b.itemID]) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// If both items have a year and they're off by more than one, it's not a dupe
|
||||
if (typeof yearCache[a.itemID] != 'undefined'
|
||||
&& typeof yearCache[b.itemID] != 'undefined'
|
||||
&& Math.abs(yearCache[a.itemID] - yearCache[b.itemID]) > 1) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Check for at least one match on last name + first initial of first name
|
||||
var aCreatorRows, bCreatorRows;
|
||||
if (typeof creatorRowsCache[a.itemID] != 'undefined') {
|
||||
aCreatorRows = creatorRowsCache[a.itemID];
|
||||
}
|
||||
if (typeof creatorRowsCache[b.itemID] != 'undefined') {
|
||||
bCreatorRows = creatorRowsCache[b.itemID];
|
||||
}
|
||||
|
||||
// Match if no creators
|
||||
if (!aCreatorRows && !bCreatorRows) {
|
||||
return 1;
|
||||
}
|
||||
|
||||
if (!aCreatorRows || !bCreatorRows) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
for (let i = 0; i < aCreatorRows.length; i++) {
|
||||
let aCreatorRow = aCreatorRows[i];
|
||||
let aLastName = aCreatorRow.lastName;
|
||||
let aFirstInitial = aCreatorRow.firstInitial || "";
|
||||
|
||||
for (let j = 0; j < bCreatorRows.length; j++) {
|
||||
let bCreatorRow = bCreatorRows[j];
|
||||
let bLastName = bCreatorRow.lastName;
|
||||
let bFirstInitial = bCreatorRow.firstInitial || "";
|
||||
|
||||
if (aLastName === bLastName && aFirstInitial === bFirstInitial) {
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}, true);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
Zotero.Duplicates.prototype._findDuplicates = async function () {
|
||||
Zotero.debug("Finding duplicates");
|
||||
|
||||
// Match on exact fields
|
||||
/*var fields = [''];
|
||||
for (let field of fields) {
|
||||
var sql = "SELECT itemID, value FROM items JOIN itemData USING (itemID) "
|
||||
+ "JOIN itemDataValues USING (valueID) "
|
||||
+ "WHERE libraryID=? AND fieldID=? "
|
||||
+ "AND itemID NOT IN (SELECT itemID FROM deletedItems) "
|
||||
+ "ORDER BY value";
|
||||
var rows = yield Zotero.DB.queryAsync(sql, [this._libraryID, Zotero.ItemFields.getID(field)]);
|
||||
processRows(rows);
|
||||
}*/
|
||||
var start = Date.now();
|
||||
|
||||
await this._loadCaches();
|
||||
|
||||
this._sets = new Zotero.DisjointSetForest;
|
||||
|
||||
this._processRows(this._isbnRows);
|
||||
this._processRows(this._doiRows);
|
||||
this._processRows(this._titleRows, Zotero.Duplicates._compareRows, true);
|
||||
|
||||
Zotero.debug("Found duplicates in " + (Date.now() - start) + " ms");
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* Build an enriched row (suitable for _compareRows) from a Zotero.Item.
|
||||
*
|
||||
* @param {Zotero.Item} item - A saved or unsaved Zotero.Item
|
||||
* @return {Object} - {itemID, value, doi?, isbn?, year?, creators?}
|
||||
*/
|
||||
Zotero.Duplicates._rowFromItem = function (item) {
|
||||
var normalizeString = Zotero.Duplicates.normalizeString;
|
||||
|
||||
var rawDOI = item.getField('DOI');
|
||||
var doi = rawDOI ? (rawDOI + '').trim().toUpperCase() : undefined;
|
||||
if (doi && !doi.startsWith('10.')) doi = undefined;
|
||||
|
||||
var rawISBN = item.getField('ISBN');
|
||||
var isbn = rawISBN ? Zotero.Utilities.cleanISBN('' + rawISBN) : undefined;
|
||||
isbn = isbn ? Zotero.Utilities.toISBN13(isbn) : undefined;
|
||||
|
||||
var year = item.getField('year') || undefined;
|
||||
|
||||
var creators = item.getCreators();
|
||||
var normalizedCreators = creators.length
|
||||
? creators.map(c => ({
|
||||
lastName: normalizeString(c.lastName || ''),
|
||||
firstInitial: c.fieldMode === 0 ? normalizeString(c.firstName || '').charAt(0) : false
|
||||
}))
|
||||
: undefined;
|
||||
|
||||
return {
|
||||
itemID: item.id || null,
|
||||
value: normalizeString(item.getField('title', false, true)),
|
||||
doi: doi,
|
||||
isbn: isbn,
|
||||
year: year,
|
||||
creators: normalizedCreators
|
||||
};
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* Find items in the library that are duplicates of the given item.
|
||||
*
|
||||
* @param {Zotero.Item|Object} itemOrCSLJSON - A Zotero.Item, or a CSL-JSON object
|
||||
* @return {Promise<Integer[]>} - Array of matching itemIDs
|
||||
*/
|
||||
Zotero.Duplicates.prototype.findDuplicatesOf = async function (itemOrCSLJSON) {
|
||||
var item;
|
||||
if (itemOrCSLJSON instanceof Zotero.Item) {
|
||||
item = itemOrCSLJSON;
|
||||
}
|
||||
else {
|
||||
item = new Zotero.Item();
|
||||
Zotero.Utilities.Item.itemFromCSLJSON(item, itemOrCSLJSON);
|
||||
}
|
||||
|
||||
await this._loadCaches();
|
||||
|
||||
var targetRow = Zotero.Duplicates._rowFromItem(item);
|
||||
var matches = new Set();
|
||||
|
||||
// ISBN exact-match pass
|
||||
if (targetRow.isbn) {
|
||||
let startIdx = _binarySearch(this._isbnRows, targetRow.isbn);
|
||||
let m = Zotero.Duplicates._checkIfDuplicate(
|
||||
{ value: targetRow.isbn },
|
||||
this._isbnRows.slice(startIdx)
|
||||
);
|
||||
for (let r of m) matches.add(r.itemID);
|
||||
}
|
||||
|
||||
// DOI exact-match pass
|
||||
if (targetRow.doi) {
|
||||
let startIdx = _binarySearch(this._doiRows, targetRow.doi);
|
||||
let m = Zotero.Duplicates._checkIfDuplicate(
|
||||
{ value: targetRow.doi },
|
||||
this._doiRows.slice(startIdx)
|
||||
);
|
||||
for (let r of m) matches.add(r.itemID);
|
||||
}
|
||||
|
||||
// Title + creators pass — reuses _compareRows directly with the enriched row
|
||||
if (targetRow.value) {
|
||||
let startIdx = _binarySearch(this._titleRows, targetRow.value);
|
||||
let m = Zotero.Duplicates._checkIfDuplicate(
|
||||
targetRow,
|
||||
this._titleRows.slice(startIdx),
|
||||
Zotero.Duplicates._compareRows
|
||||
);
|
||||
for (let r of m) matches.add(r.itemID);
|
||||
}
|
||||
|
||||
// Filter out the target item itself if it was a library item
|
||||
if (targetRow.itemID) matches.delete(targetRow.itemID);
|
||||
return [...matches];
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* Binary search for the first row whose value >= the target value
|
||||
* in a sorted rows array.
|
||||
*
|
||||
* @param {Object[]} rows - Sorted by .value
|
||||
* @param {String} value - Target value to find
|
||||
* @return {Integer} - Index of first row with value >= target
|
||||
*/
|
||||
function _binarySearch(rows, value) {
|
||||
let lo = 0, hi = rows.length;
|
||||
while (lo < hi) {
|
||||
let mid = (lo + hi) >> 1;
|
||||
if (rows[mid].value < value) {
|
||||
lo = mid + 1;
|
||||
}
|
||||
else {
|
||||
hi = mid;
|
||||
}
|
||||
}
|
||||
return lo;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Implements the Disjoint Set data structure
|
||||
|
|
|
|||
|
|
@ -60,6 +60,158 @@ describe("Duplicate Items", function () {
|
|||
await waitForNotifierEvent('refresh', 'trash');
|
||||
}
|
||||
|
||||
describe("findDuplicatesOf()", function () {
|
||||
it("should find duplicates of a Zotero.Item by title + creator", async function () {
|
||||
var item1 = await createDataObject('item', {
|
||||
title: 'Test Dedup Title',
|
||||
creators: [{
|
||||
firstName: 'John',
|
||||
lastName: 'Smith',
|
||||
creatorType: 'author'
|
||||
}]
|
||||
});
|
||||
var item2 = await createDataObject('item', {
|
||||
title: 'Test Dedup Title',
|
||||
creators: [{
|
||||
firstName: 'John',
|
||||
lastName: 'Smith',
|
||||
creatorType: 'author'
|
||||
}]
|
||||
});
|
||||
// Different title, should not match
|
||||
var item3 = await createDataObject('item', {
|
||||
title: 'Different Title',
|
||||
creators: [{
|
||||
firstName: 'John',
|
||||
lastName: 'Smith',
|
||||
creatorType: 'author'
|
||||
}]
|
||||
});
|
||||
|
||||
var d = new Zotero.Duplicates(Zotero.Libraries.userLibraryID);
|
||||
var dupes = await d.findDuplicatesOf(item1);
|
||||
assert.include(dupes, item2.id);
|
||||
assert.notInclude(dupes, item1.id);
|
||||
assert.notInclude(dupes, item3.id);
|
||||
});
|
||||
|
||||
it("should find duplicates of a CSL-JSON item by title + creator", async function () {
|
||||
var item1 = await createDataObject('item', {
|
||||
title: 'CSL Dedup Title',
|
||||
creators: [{
|
||||
firstName: 'Jane',
|
||||
lastName: 'Doe',
|
||||
creatorType: 'author'
|
||||
}]
|
||||
});
|
||||
|
||||
var cslItem = {
|
||||
type: 'book',
|
||||
title: 'CSL Dedup Title',
|
||||
author: [{ family: 'Doe', given: 'Jane' }]
|
||||
};
|
||||
|
||||
var d = new Zotero.Duplicates(Zotero.Libraries.userLibraryID);
|
||||
var dupes = await d.findDuplicatesOf(cslItem);
|
||||
assert.include(dupes, item1.id);
|
||||
});
|
||||
|
||||
it("should find duplicates by DOI", async function () {
|
||||
var item1 = await createDataObject('item', {
|
||||
itemType: 'journalArticle',
|
||||
title: 'Article One'
|
||||
});
|
||||
item1.setField('DOI', '10.1234/test.doi');
|
||||
await item1.saveTx();
|
||||
|
||||
var cslItem = {
|
||||
type: 'article-journal',
|
||||
title: 'Completely Different Title',
|
||||
DOI: '10.1234/test.doi'
|
||||
};
|
||||
|
||||
var d = new Zotero.Duplicates(Zotero.Libraries.userLibraryID);
|
||||
var dupes = await d.findDuplicatesOf(cslItem);
|
||||
assert.include(dupes, item1.id);
|
||||
});
|
||||
|
||||
it("should find duplicates by ISBN", async function () {
|
||||
var item1 = await createDataObject('item', {
|
||||
itemType: 'book',
|
||||
title: 'My Book'
|
||||
});
|
||||
item1.setField('ISBN', '978-0-306-40615-7');
|
||||
await item1.saveTx();
|
||||
|
||||
var cslItem = {
|
||||
type: 'book',
|
||||
title: 'Some Other Book Title',
|
||||
ISBN: '978-0-306-40615-7'
|
||||
};
|
||||
|
||||
var d = new Zotero.Duplicates(Zotero.Libraries.userLibraryID);
|
||||
var dupes = await d.findDuplicatesOf(cslItem);
|
||||
assert.include(dupes, item1.id);
|
||||
});
|
||||
|
||||
it("should not match items with same title but conflicting years", async function () {
|
||||
var item1 = await createDataObject('item', {
|
||||
title: 'Year Conflict Title',
|
||||
creators: [{
|
||||
firstName: 'Alice',
|
||||
lastName: 'Test',
|
||||
creatorType: 'author'
|
||||
}]
|
||||
});
|
||||
item1.setField('date', '2020');
|
||||
await item1.saveTx();
|
||||
|
||||
var cslItem = {
|
||||
type: 'book',
|
||||
title: 'Year Conflict Title',
|
||||
author: [{ family: 'Test', given: 'Alice' }],
|
||||
issued: { 'date-parts': [[2015]] }
|
||||
};
|
||||
|
||||
var d = new Zotero.Duplicates(Zotero.Libraries.userLibraryID);
|
||||
var dupes = await d.findDuplicatesOf(cslItem);
|
||||
assert.notInclude(dupes, item1.id);
|
||||
});
|
||||
|
||||
it("should not match items with same title but different creators", async function () {
|
||||
var item1 = await createDataObject('item', {
|
||||
title: 'Creator Mismatch Title',
|
||||
creators: [{
|
||||
firstName: 'Alice',
|
||||
lastName: 'One',
|
||||
creatorType: 'author'
|
||||
}]
|
||||
});
|
||||
|
||||
var cslItem = {
|
||||
type: 'book',
|
||||
title: 'Creator Mismatch Title',
|
||||
author: [{ family: 'Two', given: 'Bob' }]
|
||||
};
|
||||
|
||||
var d = new Zotero.Duplicates(Zotero.Libraries.userLibraryID);
|
||||
var dupes = await d.findDuplicatesOf(cslItem);
|
||||
assert.notInclude(dupes, item1.id);
|
||||
});
|
||||
|
||||
it("should return empty array when no duplicates exist", async function () {
|
||||
var cslItem = {
|
||||
type: 'book',
|
||||
title: 'Absolutely Unique Title ' + Zotero.Utilities.randomString(),
|
||||
author: [{ family: 'Nobody', given: 'X' }]
|
||||
};
|
||||
|
||||
var d = new Zotero.Duplicates(Zotero.Libraries.userLibraryID);
|
||||
var dupes = await d.findDuplicatesOf(cslItem);
|
||||
assert.lengthOf(dupes, 0);
|
||||
});
|
||||
});
|
||||
|
||||
describe("Merging", function () {
|
||||
it("should merge two items in duplicates view", async function () {
|
||||
var item1 = await createDataObject('item', { setTitle: true });
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue