'use strict'; import {fuzzyMatch} from './utils.js'; const RRF_K = 60; /** Merges independently-ranked result lists without comparing their raw Xapian scores. */ export function rrfMerge(groups) { const merged = new Map(); for(const group of groups) { group.forEach((hit, rank) => { const contribution = 1 / (RRF_K + rank + 1); const entry = merged.get(hit.href); if(entry) entry.score += contribution; else merged.set(hit.href, {hit, score: contribution}); }); } return [...merged.values()] .sort((a, b) => b.score - a.score) .map(({hit, score}) => ({...hit, rrf: score})); } /** * Calculates field relevance from exact terms, phrase matches, proximity, * match density and optionally fuzzy similarity. */ function fieldScore(text, terms, {fuzzy = false} = {}) { const value = String(text || '').trim().toLowerCase(); if(!value || !terms.length) { return { exact: 0, coverage: 0, proximity: 0, density: 0, fuzzy: 0, phrase: 0, score: 0, }; } const words = value.split(/\W+/).filter(Boolean); const normalizedTerms = terms.map(t => t.toLowerCase()); const phrase = normalizedTerms.join(' '); const exactTerms = normalizedTerms.filter(t => words.includes(t)); const substringTerms = normalizedTerms.filter(t => value.includes(t)); const coverage = exactTerms.length / normalizedTerms.length; const substringCoverage = substringTerms.length / normalizedTerms.length; const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage; const phraseScore = value.includes(phrase) ? 1 : 0; let proximity = 0; if(normalizedTerms.length > 1) { const positions = []; for(const term of normalizedTerms) { const index = words.indexOf(term); if(index !== -1) positions.push(index); } if(positions.length > 1) { const span = Math.max(...positions) - Math.min(...positions); proximity = 1 / Math.max(1, span); } } const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0); const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25)); const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0; const score = phraseScore * 1 + exact * 0.8 + proximity * 0.35 + substringCoverage * 0.25 + density * 0.15 + fuzzyScore * 0.75; return { exact, coverage, proximity, density, fuzzy: fuzzyScore, phrase: phraseScore, score, }; } /** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */ export function rerank(hits, termList) { if(!termList.length) return hits; return hits.map(hit => { const title = fieldScore(hit.title, termList, {fuzzy: true}); const summary = fieldScore(hit.summary, termList); const titleLower = (hit.title || '').toLowerCase(); const summaryLower = (hit.summary?._text || '').toLowerCase(); const phrase = termList.join(' ').toLowerCase(); const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0; const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0; const allTitleTerms = termList.every(term => titleLower.split(/\W+/).includes(term.toLowerCase()) ) ? 1 : 0; const allSummaryTerms = termList.every(term => summaryLower.includes(term.toLowerCase()) ) ? 1 : 0; const finalScore = hit.rrf + title.score * 1.25 + summary.score * 0.35 + titleExactPhrase * 1.5 + allTitleTerms * 0.75 + summaryExactPhrase * 0.2 + allSummaryTerms * 0.15; return { ...hit, finalScore, ranking: { rrf: hit.rrf, title: title.score, summary: summary.score, phrase: titleExactPhrase, coverage: title.coverage, proximity: title.proximity, fuzzy: title.fuzzy, }, }; }).sort((a, b) => b.finalScore - a.finalScore); } /** Round-robins results between source ZIMs so one archive cannot dominate the page. */ export function diversify(hits, limit) { const byBook = new Map(); for(const hit of hits) (byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit); for(const list of byBook.values()) list.sort((a, b) => b.finalScore - a.finalScore); const queues = [...byBook.values()]; const out = []; for(let i = 0; out.length < limit && queues.some(q => q.length); i++) { const queue = queues[i % queues.length]; if(queue.length) out.push(queue.shift()); } return out; } function bucketByFirstChar(words) { const buckets = new Map(); for(const word of words) { const key = word[0]; (buckets.get(key) ?? buckets.set(key, []).get(key)).push(word); } return buckets; } /** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */ export function bestFuzzyMatch(term, buckets) { const lower = term.toLowerCase(); const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34)); const candidates = new Set(); // Check every character so a typo in the first character doesn't eliminate the correct word. for(const char of lower) { const bucket = buckets.get(char); if(bucket) for(const candidate of bucket) candidates.add(candidate); } let best = null; let bestScore = 0; for(const candidate of candidates) { if(Math.abs(candidate.length - lower.length) > maxDistance) continue; const score = fuzzyMatch(candidate, lower).max; if(score > bestScore) { bestScore = score; best = candidate; } } return best; } /** Suggests corrected terms from the supplied title vocabulary. */ export function suggestCorrection(termList, vocabulary) { if(!vocabulary.size) return null; const buckets = bucketByFirstChar(vocabulary); let changed = false; const corrected = termList.map(term => { if(vocabulary.has(term.toLowerCase())) return term; const fix = bestFuzzyMatch(term, buckets); if(fix && fix !== term.toLowerCase()) { changed = true; return fix; } return term; }); return changed ? corrected : null; }