diff --git a/package.json b/package.json index ccb9107..eba9cb8 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@ztimson/zim-utils", - "version": "0.3.8", + "version": "0.4.0", "description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js", "author": "Zak Timson", "license": "MIT", diff --git a/src/manager.js b/src/manager.js index 4c5cc65..9d77472 100644 --- a/src/manager.js +++ b/src/manager.js @@ -107,7 +107,7 @@ export class ZimManager { const file = match.href + (match.href.endsWith('.zim') ? '' : '.zim'); await fs.promises.rm(path.join(this.#dir, file), {force: true}); await this.#ensureServer(); - await this.#waitUntilVisible(file, {expect: false}); + await this.#waitUntilVicsible(file, {expect: false}); return {file: match.file, status: 'deleted'}; } @@ -154,9 +154,9 @@ export class ZimManager { } /** Two-pass fulltext search across every local ZIM, returns enriched results matching catalog/list shape. */ - async search(terms, limit = 20) { + async search(terms, {limit = 20, sources = null} = {}) { const server = await this.#ensureServer(); - return server.search(terms, limit); + return server.search(terms, {limit, sources}); } /** Checks all local ZIMs against the catalog and updates any that are outdated. */ diff --git a/src/reader.js b/src/reader.js index 81ecd4f..e94adef 100644 --- a/src/reader.js +++ b/src/reader.js @@ -6,6 +6,8 @@ import {decompressPool} from './decompress.js'; const HEADER_SIZE = 80; const NS_CONTENT = 'C'; const NS_METADATA = 'M'; +const VOCAB_CACHE_LIMIT = 4; +const vocabCache = new Map(); // zimPath -> {mtimeMs, size, words: Set}; async function readAt(fd, pos, length) { const buf = Buffer.alloc(length); @@ -33,7 +35,10 @@ async function readMimeTypes(fd, mimeListPos) { for (;;) { str += (await readAt(fd, pos, 1024)).toString('binary'); const end = str.indexOf('\0\0'); - if (end !== -1) { str = str.slice(0, end + 1); break; } + if (end !== -1) { + str = str.slice(0, end + 1); + break; + } pos += 1024; } return str.split('\0').filter(Boolean); @@ -50,8 +55,15 @@ async function readDirent(fd, offset) { o += 4; // revision, unused let redirectIndex = null, cluster = null, blob = null; - if (mimetype === 0xffff) { redirectIndex = buf.readUInt32LE(o); o += 4; } - else { cluster = buf.readUInt32LE(o); o += 4; blob = buf.readUInt32LE(o); o += 4; } + if (mimetype === 0xffff) { + redirectIndex = buf.readUInt32LE(o); + o += 4; + } else { + cluster = buf.readUInt32LE(o); + o += 4; + blob = buf.readUInt32LE(o); + o += 4; + } const urlEnd = buf.indexOf(0, o); if (urlEnd === -1) continue; @@ -73,7 +85,8 @@ async function findByUrl(fd, header, url, namespace) { const dirKey = dirent.namespace + dirent.url; const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0; if (cmp === 0) return dirent; - if (cmp < 0) hi = mid - 1; else lo = mid + 1; + if (cmp < 0) hi = mid - 1; + else lo = mid + 1; } return null; } @@ -136,3 +149,37 @@ export async function readZimMetadata(zimPath) { author: creator, publisher, }; } + +/** + * Unique, lowercased words pulled from every article title in a ZIM. + * Only the deduped Set is cached (not the raw title list), and only for the + * last VOCAB_CACHE_LIMIT ZIMs touched (evicted least-recently-used), so memory scales with recent search activity rather than total library size. + * First call per ZIM does a full O(articleCount) directory walk; cached after that until the file's mtime/size changes. + */ +export async function titleVocabulary(zimPath) { + const stat = await fs.promises.stat(zimPath); + const cached = vocabCache.get(zimPath); + if (cached && cached.mtimeMs === stat.mtimeMs && cached.size === stat.size) { + vocabCache.delete(zimPath); + vocabCache.set(zimPath, cached); // bump to most-recently-used + return cached.words; + } + + const fd = await fs.promises.open(zimPath, 'r'); + let words; + try { + const header = await readHeader(fd); + words = new Set(); + for (let i = 0; i < header.articleCount; i++) { + const dirent = await readDirent(fd, await ptr64(fd, header.urlPtrPos, i)); + if (dirent.namespace !== NS_CONTENT || dirent.mimetype === 0xffff) continue; // skip redirects/non-content + for (const w of (dirent.title || dirent.url).toLowerCase().split(/\W+/)) if (w) words.add(w); + } + } finally { + await fd.close(); + } + + vocabCache.set(zimPath, {mtimeMs: stat.mtimeMs, size: stat.size, words}); + if (vocabCache.size > VOCAB_CACHE_LIMIT) vocabCache.delete(vocabCache.keys().next().value); + return words; +} diff --git a/src/search.js b/src/search.js new file mode 100644 index 0000000..419a4b3 --- /dev/null +++ b/src/search.js @@ -0,0 +1,225 @@ +'use strict'; + +import {fuzzyMatch} from './utils.js'; + +const RRF_K = 60; + +/** Merges independently-ranked result lists without comparing their raw Xapian scores. */ +export function rrfMerge(groups) { + const merged = new Map(); + + for(const group of groups) { + group.forEach((hit, rank) => { + const contribution = 1 / (RRF_K + rank + 1); + const entry = merged.get(hit.href); + + if(entry) entry.score += contribution; + else merged.set(hit.href, {hit, score: contribution}); + }); + } + + return [...merged.values()] + .sort((a, b) => b.score - a.score) + .map(({hit, score}) => ({...hit, rrf: score})); +} + +/** + * Calculates field relevance from exact terms, phrase matches, proximity, + * match density and optionally fuzzy similarity. + */ +function fieldScore(text, terms, {fuzzy = false} = {}) { + const value = String(text || '').trim().toLowerCase(); + + if(!value || !terms.length) { + return { + exact: 0, + coverage: 0, + proximity: 0, + density: 0, + fuzzy: 0, + phrase: 0, + score: 0, + }; + } + + const words = value.split(/\W+/).filter(Boolean); + const normalizedTerms = terms.map(t => t.toLowerCase()); + const phrase = normalizedTerms.join(' '); + + const exactTerms = normalizedTerms.filter(t => words.includes(t)); + const substringTerms = normalizedTerms.filter(t => value.includes(t)); + + const coverage = exactTerms.length / normalizedTerms.length; + const substringCoverage = substringTerms.length / normalizedTerms.length; + const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage; + const phraseScore = value.includes(phrase) ? 1 : 0; + + let proximity = 0; + + if(normalizedTerms.length > 1) { + const positions = []; + + for(const term of normalizedTerms) { + const index = words.indexOf(term); + if(index !== -1) positions.push(index); + } + + if(positions.length > 1) { + const span = Math.max(...positions) - Math.min(...positions); + proximity = 1 / Math.max(1, span); + } + } + + const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0); + const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25)); + const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0; + + const score = + phraseScore * 1 + + exact * 0.8 + + proximity * 0.35 + + substringCoverage * 0.25 + + density * 0.15 + + fuzzyScore * 0.75; + + return { + exact, + coverage, + proximity, + density, + fuzzy: fuzzyScore, + phrase: phraseScore, + score, + }; +} + +/** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */ +export function rerank(hits, termList) { + if(!termList.length) return hits; + + return hits.map(hit => { + const title = fieldScore(hit.title, termList, {fuzzy: true}); + const summary = fieldScore(hit.summary, termList); + const titleLower = (hit.title || '').toLowerCase(); + const summaryLower = (hit.summary || '').toLowerCase(); + const phrase = termList.join(' ').toLowerCase(); + + const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0; + const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0; + + const allTitleTerms = termList.every(term => + titleLower.split(/\W+/).includes(term.toLowerCase()) + ) ? 1 : 0; + + const allSummaryTerms = termList.every(term => + summaryLower.includes(term.toLowerCase()) + ) ? 1 : 0; + + const finalScore = + hit.rrf + + title.score * 1.25 + + summary.score * 0.35 + + titleExactPhrase * 1.5 + + allTitleTerms * 0.75 + + summaryExactPhrase * 0.2 + + allSummaryTerms * 0.15; + + return { + ...hit, + finalScore, + ranking: { + rrf: hit.rrf, + title: title.score, + summary: summary.score, + phrase: titleExactPhrase, + coverage: title.coverage, + proximity: title.proximity, + fuzzy: title.fuzzy, + }, + }; + }).sort((a, b) => b.finalScore - a.finalScore); +} + +/** Round-robins results between source ZIMs so one archive cannot dominate the page. */ +export function diversify(hits, limit) { + const byBook = new Map(); + + for(const hit of hits) + (byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit); + + for(const list of byBook.values()) + list.sort((a, b) => b.finalScore - a.finalScore); + + const queues = [...byBook.values()]; + const out = []; + + for(let i = 0; out.length < limit && queues.some(q => q.length); i++) { + const queue = queues[i % queues.length]; + if(queue.length) out.push(queue.shift()); + } + + return out; +} + +function bucketByFirstChar(words) { + const buckets = new Map(); + + for(const word of words) { + const key = word[0]; + (buckets.get(key) ?? buckets.set(key, []).get(key)).push(word); + } + + return buckets; +} + +/** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */ +export function bestFuzzyMatch(term, buckets) { + const lower = term.toLowerCase(); + const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34)); + const candidates = new Set(); + + // Check every character so a typo in the first character doesn't eliminate the correct word. + for(const char of lower) { + const bucket = buckets.get(char); + if(bucket) for(const candidate of bucket) candidates.add(candidate); + } + + let best = null; + let bestScore = 0; + + for(const candidate of candidates) { + if(Math.abs(candidate.length - lower.length) > maxDistance) continue; + + const score = fuzzyMatch(candidate, lower).max; + + if(score > bestScore) { + bestScore = score; + best = candidate; + } + } + + return best; +} + +/** Suggests corrected terms from the supplied title vocabulary. */ +export function suggestCorrection(termList, vocabulary) { + if(!vocabulary.size) return null; + + const buckets = bucketByFirstChar(vocabulary); + let changed = false; + + const corrected = termList.map(term => { + if(vocabulary.has(term.toLowerCase())) return term; + + const fix = bestFuzzyMatch(term, buckets); + + if(fix && fix !== term.toLowerCase()) { + changed = true; + return fix; + } + + return term; + }); + + return changed ? corrected : null; +} diff --git a/src/server.js b/src/server.js index b3fd16c..62bac2b 100644 --- a/src/server.js +++ b/src/server.js @@ -7,10 +7,12 @@ import fs from 'node:fs'; import path from 'node:path'; import {fileURLToPath} from 'node:url'; import {fromXml, makeArray} from '@ztimson/utils'; +import {titleVocabulary} from './reader.js'; +import {diversify, rerank, rrfMerge, suggestCorrection} from './search.js'; const execFileAsync = promisify(execFile); const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin'); // npm package root/bin - where bin/install.js drops the kiwix-tools binaries +const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin'); const READY_TIMEOUT = 10_000; const READY_POLL_INTERVAL = 100; const WATCH_DEBOUNCE = 300; @@ -26,9 +28,10 @@ function findFreePort() { }); } -/** Owns a kiwix-serve process's full lifecycle: library.xml, start/stop/reload, content + search access. - * Watches its own directory for .zim files appearing/disappearing (from *any* writer - itself, a remote-attached - * ZimManager, a script, whatever) and keeps library.xml + the running kiwix-serve process in sync automatically. */ +function tokenize(terms) { + return String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean); +} + export class KiwixServer { static #empty = '\n\n'; @@ -38,7 +41,7 @@ export class KiwixServer { #binDir; #libraryPath; #child = null; - #remote; // baseUrl string if attached to an externally-managed kiwix-serve, else null + #remote; #watcher = null; #watchTimer = null; @@ -46,11 +49,6 @@ export class KiwixServer { get running() { return !!this.#remote || !!this.#child; } get baseUrl() { return this.#remote || (this.#child ? `http://${this.#host}:${this.#port}` : null); } - /** @param {{port?: number, host?: string, binDir?: string, url?: string}} [opts] - * url: attach to an already-running kiwix-serve (e.g. one started elsewhere in your codebase) instead of - * spawning/owning one - start/stop become no-ops, and library.xml is read over HTTP instead of disk. - * Whichever instance *does* own the process is responsible for watching the shared directory; an attached - * instance doesn't need its own watcher; it just reads whatever the owner already reloaded. */ constructor(dir, {port, host = '127.0.0.1', binDir = DEFAULT_BIN_DIR, url} = {}) { this.#dir = dir; this.#host = host; @@ -58,25 +56,29 @@ export class KiwixServer { this.#binDir = binDir; this.#libraryPath = path.join(dir, 'library.xml'); this.#remote = url ? url.replace(/\/$/, '') : null; - if (!this.#remote) this.#ensureLocalStore(); + + if(!this.#remote) this.#ensureLocalStore(); } #ensureLocalStore() { fs.mkdirSync(this.#dir, {recursive: true}); - if (!fs.existsSync(this.#libraryPath)) fs.writeFileSync(this.#libraryPath, KiwixServer.#empty); + + if(!fs.existsSync(this.#libraryPath)) + fs.writeFileSync(this.#libraryPath, KiwixServer.#empty); } #assertRunning() { - if (!this.running) throw new Error('KiwixServer is not running - call start() first'); + if(!this.running) throw new Error('KiwixServer is not running - call start() first'); } #bin(name) { return path.join(this.#binDir, process.platform === 'win32' ? `${name}.exe` : name); } - /** Reads library.xml from disk if we own the server, or over HTTP if attached to a remote one. */ async #fetchLibraryXml() { - if (!this.#remote) return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => ''); + if(!this.#remote) + return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => ''); + try { const res = await fetch(`${this.#remote}/library.xml`); return res.ok ? await res.text() : ''; @@ -85,44 +87,55 @@ export class KiwixServer { } } - /** Rebuilds library.xml from scratch by scanning `dir` for .zim files - no-op if attached to a remote server. - * Rewriting the file bumps its mtime, which kiwix-serve (started with --monitorLibrary) picks up on its own - * and hot-reloads without needing a restart. */ async #rebuildLibrary() { - if (this.#remote) return; + if(this.#remote) return; + await fs.promises.rm(this.#libraryPath, {force: true}); + const files = await this.#zimFiles(); - if (!files.length) return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty); - for (const f of files) await execFileAsync(this.#bin('kiwix-manage'), [this.#libraryPath, 'add', path.join(this.#dir, f)]); + + if(!files.length) + return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty); + + for(const file of files) + await execFileAsync(this.#bin('kiwix-manage'), [ + this.#libraryPath, + 'add', + path.join(this.#dir, file), + ]); } async #waitUntilReady() { const deadline = Date.now() + READY_TIMEOUT; - while (Date.now() < deadline) { + + while(Date.now() < deadline) { try { await fetch(`http://${this.#host}:${this.#port}/`); return; } catch {} + await new Promise(r => setTimeout(r, READY_POLL_INTERVAL)); } + throw new Error('kiwix-serve did not become ready in time'); } async #zimFiles() { - return (await fs.promises.readdir(this.#dir).catch(() => [])).filter(f => f.endsWith('.zim')); + return (await fs.promises.readdir(this.#dir).catch(() => [])) + .filter(f => f.endsWith('.zim')); } - /** Watches `dir` for .zim files being added/removed by anyone (this process, a remote-attached ZimManager, - * a manual copy) and debounces a library rebuild so kiwix-serve's --monitorLibrary picks it up. Only runs - * while we own the process - an attached instance has nothing local to watch on behalf of. */ #watchDir() { this.#watcher?.close(); + this.#watcher = fs.watch(this.#dir, (_event, filename) => { - if (!filename?.endsWith('.zim')) return; + if(!filename?.endsWith('.zim')) return; + clearTimeout(this.#watchTimer); this.#watchTimer = setTimeout(() => this.#rebuildLibrary().catch(() => {}), WATCH_DEBOUNCE); }); - this.#watcher.on('error', () => {}); // e.g. dir removed out from under us - just stop watching, don't crash the process + + this.#watcher.on('error', () => {}); } #unwatchDir() { @@ -132,23 +145,31 @@ export class KiwixServer { this.#watcher = null; } - /** Rebuilds library.xml and starts kiwix-serve. Resolves once the server is responding. - * Starts with --monitorLibrary so the process reloads itself whenever library.xml's mtime changes - - * no restart needed for updates after this. */ async start() { - if (this.#remote || this.#child) return; + if(this.#remote || this.#child) return; + await fs.promises.mkdir(this.#dir, {recursive: true}); await this.#rebuildLibrary(); this.#port ??= await findFreePort(); - this.#child = spawn(this.#bin('kiwix-serve'), - ['--library', '--monitorLibrary', '-i', this.#host, '-p', String(this.#port), this.#libraryPath], - {stdio: 'ignore'}); - this.#child.on('exit', () => { this.#child = null; this.#unwatchDir(); }); + this.#child = spawn(this.#bin('kiwix-serve'), [ + '--library', + '--monitorLibrary', + '-i', + this.#host, + '-p', + String(this.#port), + this.#libraryPath, + ], {stdio: 'ignore'}); + + this.#child.on('exit', () => { + this.#child = null; + this.#unwatchDir(); + }); try { await this.#waitUntilReady(); - } catch (e) { + } catch(e) { await this.stop(); throw e; } @@ -156,15 +177,18 @@ export class KiwixServer { this.#watchDir(); } - /** Gracefully stops kiwix-serve, if we own it. No-op if attached to a remote instance. */ async stop() { this.#unwatchDir(); - if (this.#remote || !this.#child) return; + + if(this.#remote || !this.#child) return; + const child = this.#child; + await new Promise(resolve => { child.once('exit', resolve); child.kill('SIGTERM'); }); + this.#child = null; } @@ -173,24 +197,23 @@ export class KiwixServer { await this.start(); } - /** Forces an immediate library rebuild rather than waiting for the directory watcher's debounce to fire. - * Not required for correctness (the watcher does this automatically for any writer), just a manual way to - * skip the ~300ms wait when you already know something changed. No-op if attached to a remote instance - - * whoever owns that process already reloads itself. */ async reload() { - if (this.#remote || !this.#child) return; + if(this.#remote || !this.#child) return; await this.#rebuildLibrary(); } - /** Local catalog listing - same flat shape as the online catalog (catalog.js), plus a `file` field. */ async list() { this.#assertRunning(); + const xml = await this.#fetchLibraryXml(); - if (!xml) return []; + if(!xml) return []; + const entries = fromXml(xml); + return makeArray(entries?.library?.book || []).map(e => { const tags = e.tags.split(';'); const name = e.path.replace('.zim', ''); + return { id: e.id, title: e.title, @@ -205,6 +228,7 @@ export class KiwixServer { publisher: e.publisher, articleCount: +e.articleCount || 0, sizeMb: +(Number(e.size) / 1024).toFixed(1) || 0, + file: e.path, href: name, icon: `data:${e.faviconMimetype || 'image/png'};base64,${e.favicon}`, viewer: `${this.baseUrl}/content/${name}`, @@ -212,58 +236,133 @@ export class KiwixServer { }); } - /** Splits a href ("zim/path/to/page") or a full content/viewer URL into {zim, path}. */ #splitHref(href) { const clean = href.replace(`${this.baseUrl}/content/`, '').replace(/^\/+/, ''); const [zim, ...rest] = clean.split('/'); return {zim, path: rest.join('/')}; } - /** Builds a kiwix-serve content URL from a href (as returned by list()/search()), or from an already-built content/viewer URL. */ link(href) { this.#assertRunning(); + const {zim, path} = this.#splitHref(href); + return `${this.baseUrl}/content/${zim}${path ? '/' + path : ''}`; } - /** Fetches a single asset's raw bytes straight from kiwix-serve. */ async raw(href) { const res = await fetch(this.link(href)); + if(!res.ok) return null; - return {mimetype: res.headers.get('content-type'), data: Buffer.from(await res.arrayBuffer())}; + + return { + mimetype: res.headers.get('content-type'), + data: Buffer.from(await res.arrayBuffer()), + }; } - /** Fulltext search across every local ZIM via kiwix-serve's own xapian index */ - async search(terms, limit = 20) { + async #rawSearch(termList, scoped, bookMap, limit) { + const groups = new Map(); + + for(const book of scoped) { + const lang = book.language || ''; + (groups.get(lang) ?? groups.set(lang, []).get(lang)).push(book); + } + + const perGroup = await Promise.all([...groups.values()].map(async group => { + const params = new URLSearchParams({ + pattern: termList.join(' '), + format: 'xml', + pageLength: String(limit), + }); + + for(const book of group) + params.append('books.name', book.name); + + const res = await fetch(`${this.baseUrl}/search?${params}`); + if(!res.ok) return []; + + const found = fromXml(await res.text())?.rss?.channel?.item || []; + + return found.map(hit => { + const book = bookMap.get(hit.book.title); + if(!book) return null; + + const prefix = `/content/${book.href}/`; + const page = hit.link.startsWith(prefix) + ? hit.link.slice(prefix.length) + : hit.link.replace(/^\/+/, ''); + + return { + id: book.id, + title: hit.title, + page, + name: book.name, + publisher: book.publisher, + href: `${book.href}/${page}`, + icon: book.icon, + viewer: this.baseUrl + hit.link, + summary: hit.description, + xapianScore: +hit.score || 0, + }; + }).filter(Boolean).sort((a, b) => b.xapianScore - a.xapianScore); + })); + + return rrfMerge(perGroup); + } + + /** Fulltext search across local ZIMs with ranking, diversification and spelling correction. */ + async search(terms, {limit = 20, sources = null} = {}) { this.#assertRunning(); - const termList = String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean); - if (!termList.length) return []; - const params = new URLSearchParams({pattern: termList.join(' '), format: 'xml', pageLength: String(limit)}); - const res = await fetch(`${this.baseUrl}/search?${params}`); - if (!res.ok) return []; + const termList = tokenize(terms); + if(!termList.length) return {results: [], spellcheck: null}; - const found = fromXml(await res.text())?.rss?.channel?.item || []; const books = await this.list(); const bookMap = new Map(books.map(b => [b.title, b])); + const scoped = sources?.length + ? books.filter(b => sources.includes(b.name) || sources.includes(b.href)) + : books; - return found.map(hit => { - const book = bookMap.get(hit.book.title); - if (!book) return null; - const prefix = `/content/${book.href}/`; - const page = hit.link.startsWith(prefix) ? hit.link.slice(prefix.length) : hit.link.replace(/^\/+/, ''); - return { - id: book.id, - title: hit.title, - page, - name: book.name, - publisher: book.publisher, - href: `${book.href}/${page}`, - icon: book.icon, - viewer: this.baseUrl + hit.link, - summary: hit.description, - score: +hit.score || 0, - }; - }).filter(Boolean); + if(!scoped.length) return {results: [], spellcheck: null}; + + let activeTerms = termList; + let hits = await this.#rawSearch(activeTerms, scoped, bookMap, limit); + let spellcheck = null; + + if(!hits.length && !this.#remote) { + const vocabulary = new Set(); + + for(const book of scoped) { + if(!book.file) continue; + + try { + for(const word of await titleVocabulary(path.join(this.#dir, book.file))) + vocabulary.add(word); + } catch { + // Unreadable ZIM - skip it, don't fail the whole search. + } + } + + const corrected = suggestCorrection(termList, vocabulary); + + if(corrected) { + const retry = await this.#rawSearch(corrected, scoped, bookMap, limit); + + if(retry.length) { + hits = retry; + activeTerms = corrected; + spellcheck = { + from: termList.join(' '), + to: corrected.join(' '), + }; + } + } + } + + return { + results: diversify(rerank(hits, activeTerms), limit), + spellcheck, + }; } } diff --git a/src/utils.js b/src/utils.js index 341c5bb..fa7c554 100644 --- a/src/utils.js +++ b/src/utils.js @@ -1,39 +1,36 @@ export function levenshtein(a, b) { const m = a.length, n = b.length; - if (!m) return n; - if (!n) return m; + if(!m) return n; + if(!n) return m; const dp = Array.from({length: m + 1}, (_, i) => [i, ...Array(n).fill(0)]); - for (let j = 0; j <= n; j++) dp[0][j] = j; - for (let i = 1; i <= m; i++) { - for (let j = 1; j <= n; j++) { - dp[i][j] = a[i - 1] === b[j - 1] - ? dp[i - 1][j - 1] - : 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]); + for(let j = 0; j <= n; j++) dp[0][j] = j; + for(let i = 1; i <= m; i++) { + for(let j = 1; j <= n; j++) { + dp[i][j] = a[i - 1] === b[j - 1] ? dp[i - 1][j - 1] : 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]); } } return dp[m][n]; } function scoreAgainst(text, term) { - if (text.includes(term)) return 1 - (text.length - term.length) / text.length * 0.3; - if (!text.length || !term.length || text[0] !== term[0]) return 0; - const dist = levenshtein(text, term); + if (!text.length || !term.length) return 0; + if (text === term) return 1; + if (text.includes(term)) return 0.8; const maxAllowed = Math.max(1, Math.ceil(term.length * 0.34)); + if (Math.abs(text.length - term.length) > maxAllowed) return 0; + const dist = levenshtein(text, term); if (dist > maxAllowed) return 0; return 1 - dist / Math.max(text.length, term.length); } -/** Compares `target` against one or more search terms; returns avg/max/per-term similarity. */ export function fuzzyMatch(target, ...terms) { if (!terms.length) throw new Error('Requires at least 1 term to compare'); const lowerTarget = String(target).toLowerCase(); const words = lowerTarget.split(/\W+/).filter(Boolean); - const similarities = terms.map(term => { - const t = term.toLowerCase(); + const t = String(term).toLowerCase(); return Math.max(scoreAgainst(lowerTarget, t), ...words.map(w => scoreAgainst(w, t))); }); - return { avg: similarities.reduce((acc, s) => acc + s, 0) / similarities.length, max: Math.max(...similarities),