generated from ztimson/template
226 lines
5.7 KiB
JavaScript
226 lines
5.7 KiB
JavaScript
'use strict';
|
|
|
|
import {fuzzyMatch} from './utils.js';
|
|
|
|
const RRF_K = 60;
|
|
|
|
/** Merges independently-ranked result lists without comparing their raw Xapian scores. */
|
|
export function rrfMerge(groups) {
|
|
const merged = new Map();
|
|
|
|
for(const group of groups) {
|
|
group.forEach((hit, rank) => {
|
|
const contribution = 1 / (RRF_K + rank + 1);
|
|
const entry = merged.get(hit.href);
|
|
|
|
if(entry) entry.score += contribution;
|
|
else merged.set(hit.href, {hit, score: contribution});
|
|
});
|
|
}
|
|
|
|
return [...merged.values()]
|
|
.sort((a, b) => b.score - a.score)
|
|
.map(({hit, score}) => ({...hit, rrf: score}));
|
|
}
|
|
|
|
/**
|
|
* Calculates field relevance from exact terms, phrase matches, proximity,
|
|
* match density and optionally fuzzy similarity.
|
|
*/
|
|
function fieldScore(text, terms, {fuzzy = false} = {}) {
|
|
const value = String(text || '').trim().toLowerCase();
|
|
|
|
if(!value || !terms.length) {
|
|
return {
|
|
exact: 0,
|
|
coverage: 0,
|
|
proximity: 0,
|
|
density: 0,
|
|
fuzzy: 0,
|
|
phrase: 0,
|
|
score: 0,
|
|
};
|
|
}
|
|
|
|
const words = value.split(/\W+/).filter(Boolean);
|
|
const normalizedTerms = terms.map(t => t.toLowerCase());
|
|
const phrase = normalizedTerms.join(' ');
|
|
|
|
const exactTerms = normalizedTerms.filter(t => words.includes(t));
|
|
const substringTerms = normalizedTerms.filter(t => value.includes(t));
|
|
|
|
const coverage = exactTerms.length / normalizedTerms.length;
|
|
const substringCoverage = substringTerms.length / normalizedTerms.length;
|
|
const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage;
|
|
const phraseScore = value.includes(phrase) ? 1 : 0;
|
|
|
|
let proximity = 0;
|
|
|
|
if(normalizedTerms.length > 1) {
|
|
const positions = [];
|
|
|
|
for(const term of normalizedTerms) {
|
|
const index = words.indexOf(term);
|
|
if(index !== -1) positions.push(index);
|
|
}
|
|
|
|
if(positions.length > 1) {
|
|
const span = Math.max(...positions) - Math.min(...positions);
|
|
proximity = 1 / Math.max(1, span);
|
|
}
|
|
}
|
|
|
|
const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0);
|
|
const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25));
|
|
const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0;
|
|
|
|
const score =
|
|
phraseScore * 1 +
|
|
exact * 0.8 +
|
|
proximity * 0.35 +
|
|
substringCoverage * 0.25 +
|
|
density * 0.15 +
|
|
fuzzyScore * 0.75;
|
|
|
|
return {
|
|
exact,
|
|
coverage,
|
|
proximity,
|
|
density,
|
|
fuzzy: fuzzyScore,
|
|
phrase: phraseScore,
|
|
score,
|
|
};
|
|
}
|
|
|
|
/** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */
|
|
export function rerank(hits, termList) {
|
|
if(!termList.length) return hits;
|
|
|
|
return hits.map(hit => {
|
|
const title = fieldScore(hit.title, termList, {fuzzy: true});
|
|
const summary = fieldScore(hit.summary, termList);
|
|
const titleLower = (hit.title || '').toLowerCase();
|
|
const summaryLower = (hit.summary?._text || '').toLowerCase();
|
|
const phrase = termList.join(' ').toLowerCase();
|
|
|
|
const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0;
|
|
const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0;
|
|
|
|
const allTitleTerms = termList.every(term =>
|
|
titleLower.split(/\W+/).includes(term.toLowerCase())
|
|
) ? 1 : 0;
|
|
|
|
const allSummaryTerms = termList.every(term =>
|
|
summaryLower.includes(term.toLowerCase())
|
|
) ? 1 : 0;
|
|
|
|
const finalScore =
|
|
hit.rrf +
|
|
title.score * 1.25 +
|
|
summary.score * 0.35 +
|
|
titleExactPhrase * 1.5 +
|
|
allTitleTerms * 0.75 +
|
|
summaryExactPhrase * 0.2 +
|
|
allSummaryTerms * 0.15;
|
|
|
|
return {
|
|
...hit,
|
|
finalScore,
|
|
ranking: {
|
|
rrf: hit.rrf,
|
|
title: title.score,
|
|
summary: summary.score,
|
|
phrase: titleExactPhrase,
|
|
coverage: title.coverage,
|
|
proximity: title.proximity,
|
|
fuzzy: title.fuzzy,
|
|
},
|
|
};
|
|
}).sort((a, b) => b.finalScore - a.finalScore);
|
|
}
|
|
|
|
/** Round-robins results between source ZIMs so one archive cannot dominate the page. */
|
|
export function diversify(hits, limit) {
|
|
const byBook = new Map();
|
|
|
|
for(const hit of hits)
|
|
(byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit);
|
|
|
|
for(const list of byBook.values())
|
|
list.sort((a, b) => b.finalScore - a.finalScore);
|
|
|
|
const queues = [...byBook.values()];
|
|
const out = [];
|
|
|
|
for(let i = 0; out.length < limit && queues.some(q => q.length); i++) {
|
|
const queue = queues[i % queues.length];
|
|
if(queue.length) out.push(queue.shift());
|
|
}
|
|
|
|
return out;
|
|
}
|
|
|
|
function bucketByFirstChar(words) {
|
|
const buckets = new Map();
|
|
|
|
for(const word of words) {
|
|
const key = word[0];
|
|
(buckets.get(key) ?? buckets.set(key, []).get(key)).push(word);
|
|
}
|
|
|
|
return buckets;
|
|
}
|
|
|
|
/** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */
|
|
export function bestFuzzyMatch(term, buckets) {
|
|
const lower = term.toLowerCase();
|
|
const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34));
|
|
const candidates = new Set();
|
|
|
|
// Check every character so a typo in the first character doesn't eliminate the correct word.
|
|
for(const char of lower) {
|
|
const bucket = buckets.get(char);
|
|
if(bucket) for(const candidate of bucket) candidates.add(candidate);
|
|
}
|
|
|
|
let best = null;
|
|
let bestScore = 0;
|
|
|
|
for(const candidate of candidates) {
|
|
if(Math.abs(candidate.length - lower.length) > maxDistance) continue;
|
|
|
|
const score = fuzzyMatch(candidate, lower).max;
|
|
|
|
if(score > bestScore) {
|
|
bestScore = score;
|
|
best = candidate;
|
|
}
|
|
}
|
|
|
|
return best;
|
|
}
|
|
|
|
/** Suggests corrected terms from the supplied title vocabulary. */
|
|
export function suggestCorrection(termList, vocabulary) {
|
|
if(!vocabulary.size) return null;
|
|
|
|
const buckets = bucketByFirstChar(vocabulary);
|
|
let changed = false;
|
|
|
|
const corrected = termList.map(term => {
|
|
if(vocabulary.has(term.toLowerCase())) return term;
|
|
|
|
const fix = bestFuzzyMatch(term, buckets);
|
|
|
|
if(fix && fix !== term.toLowerCase()) {
|
|
changed = true;
|
|
return fix;
|
|
}
|
|
|
|
return term;
|
|
});
|
|
|
|
return changed ? corrected : null;
|
|
}
|