Files
zim-utils/src/search.js
T
ztimson 0a7e87f6b7
Publish Library / Build NPM Project (push) Successful in 3m16s
Publish Library / Tag Version (push) Successful in 19s
Patched search results
2026-09-20 20:29:28 -04:00

226 lines
5.7 KiB
JavaScript

'use strict';
import {fuzzyMatch} from './utils.js';
const RRF_K = 60;
/** Merges independently-ranked result lists without comparing their raw Xapian scores. */
export function rrfMerge(groups) {
const merged = new Map();
for(const group of groups) {
group.forEach((hit, rank) => {
const contribution = 1 / (RRF_K + rank + 1);
const entry = merged.get(hit.href);
if(entry) entry.score += contribution;
else merged.set(hit.href, {hit, score: contribution});
});
}
return [...merged.values()]
.sort((a, b) => b.score - a.score)
.map(({hit, score}) => ({...hit, rrf: score}));
}
/**
* Calculates field relevance from exact terms, phrase matches, proximity,
* match density and optionally fuzzy similarity.
*/
function fieldScore(text, terms, {fuzzy = false} = {}) {
const value = String(text || '').trim().toLowerCase();
if(!value || !terms.length) {
return {
exact: 0,
coverage: 0,
proximity: 0,
density: 0,
fuzzy: 0,
phrase: 0,
score: 0,
};
}
const words = value.split(/\W+/).filter(Boolean);
const normalizedTerms = terms.map(t => t.toLowerCase());
const phrase = normalizedTerms.join(' ');
const exactTerms = normalizedTerms.filter(t => words.includes(t));
const substringTerms = normalizedTerms.filter(t => value.includes(t));
const coverage = exactTerms.length / normalizedTerms.length;
const substringCoverage = substringTerms.length / normalizedTerms.length;
const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage;
const phraseScore = value.includes(phrase) ? 1 : 0;
let proximity = 0;
if(normalizedTerms.length > 1) {
const positions = [];
for(const term of normalizedTerms) {
const index = words.indexOf(term);
if(index !== -1) positions.push(index);
}
if(positions.length > 1) {
const span = Math.max(...positions) - Math.min(...positions);
proximity = 1 / Math.max(1, span);
}
}
const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0);
const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25));
const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0;
const score =
phraseScore * 1 +
exact * 0.8 +
proximity * 0.35 +
substringCoverage * 0.25 +
density * 0.15 +
fuzzyScore * 0.75;
return {
exact,
coverage,
proximity,
density,
fuzzy: fuzzyScore,
phrase: phraseScore,
score,
};
}
/** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */
export function rerank(hits, termList) {
if(!termList.length) return hits;
return hits.map(hit => {
const title = fieldScore(hit.title, termList, {fuzzy: true});
const summary = fieldScore(hit.summary, termList);
const titleLower = (hit.title || '').toLowerCase();
const summaryLower = (hit.summary?._text || '').toLowerCase();
const phrase = termList.join(' ').toLowerCase();
const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0;
const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0;
const allTitleTerms = termList.every(term =>
titleLower.split(/\W+/).includes(term.toLowerCase())
) ? 1 : 0;
const allSummaryTerms = termList.every(term =>
summaryLower.includes(term.toLowerCase())
) ? 1 : 0;
const finalScore =
hit.rrf +
title.score * 1.25 +
summary.score * 0.35 +
titleExactPhrase * 1.5 +
allTitleTerms * 0.75 +
summaryExactPhrase * 0.2 +
allSummaryTerms * 0.15;
return {
...hit,
finalScore,
ranking: {
rrf: hit.rrf,
title: title.score,
summary: summary.score,
phrase: titleExactPhrase,
coverage: title.coverage,
proximity: title.proximity,
fuzzy: title.fuzzy,
},
};
}).sort((a, b) => b.finalScore - a.finalScore);
}
/** Round-robins results between source ZIMs so one archive cannot dominate the page. */
export function diversify(hits, limit) {
const byBook = new Map();
for(const hit of hits)
(byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit);
for(const list of byBook.values())
list.sort((a, b) => b.finalScore - a.finalScore);
const queues = [...byBook.values()];
const out = [];
for(let i = 0; out.length < limit && queues.some(q => q.length); i++) {
const queue = queues[i % queues.length];
if(queue.length) out.push(queue.shift());
}
return out;
}
function bucketByFirstChar(words) {
const buckets = new Map();
for(const word of words) {
const key = word[0];
(buckets.get(key) ?? buckets.set(key, []).get(key)).push(word);
}
return buckets;
}
/** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */
export function bestFuzzyMatch(term, buckets) {
const lower = term.toLowerCase();
const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34));
const candidates = new Set();
// Check every character so a typo in the first character doesn't eliminate the correct word.
for(const char of lower) {
const bucket = buckets.get(char);
if(bucket) for(const candidate of bucket) candidates.add(candidate);
}
let best = null;
let bestScore = 0;
for(const candidate of candidates) {
if(Math.abs(candidate.length - lower.length) > maxDistance) continue;
const score = fuzzyMatch(candidate, lower).max;
if(score > bestScore) {
bestScore = score;
best = candidate;
}
}
return best;
}
/** Suggests corrected terms from the supplied title vocabulary. */
export function suggestCorrection(termList, vocabulary) {
if(!vocabulary.size) return null;
const buckets = bucketByFirstChar(vocabulary);
let changed = false;
const corrected = termList.map(term => {
if(vocabulary.has(term.toLowerCase())) return term;
const fix = bestFuzzyMatch(term, buckets);
if(fix && fix !== term.toLowerCase()) {
changed = true;
return fix;
}
return term;
});
return changed ? corrected : null;
}