generated from ztimson/template
Better search results
This commit is contained in:
+225
@@ -0,0 +1,225 @@
|
||||
'use strict';
|
||||
|
||||
import {fuzzyMatch} from './utils.js';
|
||||
|
||||
const RRF_K = 60;
|
||||
|
||||
/** Merges independently-ranked result lists without comparing their raw Xapian scores. */
|
||||
export function rrfMerge(groups) {
|
||||
const merged = new Map();
|
||||
|
||||
for(const group of groups) {
|
||||
group.forEach((hit, rank) => {
|
||||
const contribution = 1 / (RRF_K + rank + 1);
|
||||
const entry = merged.get(hit.href);
|
||||
|
||||
if(entry) entry.score += contribution;
|
||||
else merged.set(hit.href, {hit, score: contribution});
|
||||
});
|
||||
}
|
||||
|
||||
return [...merged.values()]
|
||||
.sort((a, b) => b.score - a.score)
|
||||
.map(({hit, score}) => ({...hit, rrf: score}));
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculates field relevance from exact terms, phrase matches, proximity,
|
||||
* match density and optionally fuzzy similarity.
|
||||
*/
|
||||
function fieldScore(text, terms, {fuzzy = false} = {}) {
|
||||
const value = String(text || '').trim().toLowerCase();
|
||||
|
||||
if(!value || !terms.length) {
|
||||
return {
|
||||
exact: 0,
|
||||
coverage: 0,
|
||||
proximity: 0,
|
||||
density: 0,
|
||||
fuzzy: 0,
|
||||
phrase: 0,
|
||||
score: 0,
|
||||
};
|
||||
}
|
||||
|
||||
const words = value.split(/\W+/).filter(Boolean);
|
||||
const normalizedTerms = terms.map(t => t.toLowerCase());
|
||||
const phrase = normalizedTerms.join(' ');
|
||||
|
||||
const exactTerms = normalizedTerms.filter(t => words.includes(t));
|
||||
const substringTerms = normalizedTerms.filter(t => value.includes(t));
|
||||
|
||||
const coverage = exactTerms.length / normalizedTerms.length;
|
||||
const substringCoverage = substringTerms.length / normalizedTerms.length;
|
||||
const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage;
|
||||
const phraseScore = value.includes(phrase) ? 1 : 0;
|
||||
|
||||
let proximity = 0;
|
||||
|
||||
if(normalizedTerms.length > 1) {
|
||||
const positions = [];
|
||||
|
||||
for(const term of normalizedTerms) {
|
||||
const index = words.indexOf(term);
|
||||
if(index !== -1) positions.push(index);
|
||||
}
|
||||
|
||||
if(positions.length > 1) {
|
||||
const span = Math.max(...positions) - Math.min(...positions);
|
||||
proximity = 1 / Math.max(1, span);
|
||||
}
|
||||
}
|
||||
|
||||
const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0);
|
||||
const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25));
|
||||
const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0;
|
||||
|
||||
const score =
|
||||
phraseScore * 1 +
|
||||
exact * 0.8 +
|
||||
proximity * 0.35 +
|
||||
substringCoverage * 0.25 +
|
||||
density * 0.15 +
|
||||
fuzzyScore * 0.75;
|
||||
|
||||
return {
|
||||
exact,
|
||||
coverage,
|
||||
proximity,
|
||||
density,
|
||||
fuzzy: fuzzyScore,
|
||||
phrase: phraseScore,
|
||||
score,
|
||||
};
|
||||
}
|
||||
|
||||
/** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */
|
||||
export function rerank(hits, termList) {
|
||||
if(!termList.length) return hits;
|
||||
|
||||
return hits.map(hit => {
|
||||
const title = fieldScore(hit.title, termList, {fuzzy: true});
|
||||
const summary = fieldScore(hit.summary, termList);
|
||||
const titleLower = (hit.title || '').toLowerCase();
|
||||
const summaryLower = (hit.summary || '').toLowerCase();
|
||||
const phrase = termList.join(' ').toLowerCase();
|
||||
|
||||
const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0;
|
||||
const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0;
|
||||
|
||||
const allTitleTerms = termList.every(term =>
|
||||
titleLower.split(/\W+/).includes(term.toLowerCase())
|
||||
) ? 1 : 0;
|
||||
|
||||
const allSummaryTerms = termList.every(term =>
|
||||
summaryLower.includes(term.toLowerCase())
|
||||
) ? 1 : 0;
|
||||
|
||||
const finalScore =
|
||||
hit.rrf +
|
||||
title.score * 1.25 +
|
||||
summary.score * 0.35 +
|
||||
titleExactPhrase * 1.5 +
|
||||
allTitleTerms * 0.75 +
|
||||
summaryExactPhrase * 0.2 +
|
||||
allSummaryTerms * 0.15;
|
||||
|
||||
return {
|
||||
...hit,
|
||||
finalScore,
|
||||
ranking: {
|
||||
rrf: hit.rrf,
|
||||
title: title.score,
|
||||
summary: summary.score,
|
||||
phrase: titleExactPhrase,
|
||||
coverage: title.coverage,
|
||||
proximity: title.proximity,
|
||||
fuzzy: title.fuzzy,
|
||||
},
|
||||
};
|
||||
}).sort((a, b) => b.finalScore - a.finalScore);
|
||||
}
|
||||
|
||||
/** Round-robins results between source ZIMs so one archive cannot dominate the page. */
|
||||
export function diversify(hits, limit) {
|
||||
const byBook = new Map();
|
||||
|
||||
for(const hit of hits)
|
||||
(byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit);
|
||||
|
||||
for(const list of byBook.values())
|
||||
list.sort((a, b) => b.finalScore - a.finalScore);
|
||||
|
||||
const queues = [...byBook.values()];
|
||||
const out = [];
|
||||
|
||||
for(let i = 0; out.length < limit && queues.some(q => q.length); i++) {
|
||||
const queue = queues[i % queues.length];
|
||||
if(queue.length) out.push(queue.shift());
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
function bucketByFirstChar(words) {
|
||||
const buckets = new Map();
|
||||
|
||||
for(const word of words) {
|
||||
const key = word[0];
|
||||
(buckets.get(key) ?? buckets.set(key, []).get(key)).push(word);
|
||||
}
|
||||
|
||||
return buckets;
|
||||
}
|
||||
|
||||
/** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */
|
||||
export function bestFuzzyMatch(term, buckets) {
|
||||
const lower = term.toLowerCase();
|
||||
const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34));
|
||||
const candidates = new Set();
|
||||
|
||||
// Check every character so a typo in the first character doesn't eliminate the correct word.
|
||||
for(const char of lower) {
|
||||
const bucket = buckets.get(char);
|
||||
if(bucket) for(const candidate of bucket) candidates.add(candidate);
|
||||
}
|
||||
|
||||
let best = null;
|
||||
let bestScore = 0;
|
||||
|
||||
for(const candidate of candidates) {
|
||||
if(Math.abs(candidate.length - lower.length) > maxDistance) continue;
|
||||
|
||||
const score = fuzzyMatch(candidate, lower).max;
|
||||
|
||||
if(score > bestScore) {
|
||||
bestScore = score;
|
||||
best = candidate;
|
||||
}
|
||||
}
|
||||
|
||||
return best;
|
||||
}
|
||||
|
||||
/** Suggests corrected terms from the supplied title vocabulary. */
|
||||
export function suggestCorrection(termList, vocabulary) {
|
||||
if(!vocabulary.size) return null;
|
||||
|
||||
const buckets = bucketByFirstChar(vocabulary);
|
||||
let changed = false;
|
||||
|
||||
const corrected = termList.map(term => {
|
||||
if(vocabulary.has(term.toLowerCase())) return term;
|
||||
|
||||
const fix = bestFuzzyMatch(term, buckets);
|
||||
|
||||
if(fix && fix !== term.toLowerCase()) {
|
||||
changed = true;
|
||||
return fix;
|
||||
}
|
||||
|
||||
return term;
|
||||
});
|
||||
|
||||
return changed ? corrected : null;
|
||||
}
|
||||
Reference in New Issue
Block a user