Compare commits

4 Commits
Author SHA1 Message Date
ztimson a45d7bad8c Search fix
Publish Library / Build NPM Project (push) Successful in 4m13s
Publish Library / Tag Version (push) Successful in 16s
2026-09-20 20:57:46 -04:00
ztimson 0a7e87f6b7 Patched search results
Publish Library / Build NPM Project (push) Successful in 3m16s
Publish Library / Tag Version (push) Successful in 19s
2026-09-20 20:29:28 -04:00
ztimson e12c3e3cc6 Fixed typo
Publish Library / Build NPM Project (push) Successful in 4m39s
Publish Library / Tag Version (push) Successful in 8s
2026-09-20 19:22:10 -04:00
ztimson 8d4258b951 Better search results
Publish Library / Build NPM Project (push) Successful in 4m47s
Publish Library / Tag Version (push) Successful in 9s
2026-09-20 19:05:25 -04:00
6 changed files with 460 additions and 99 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@ztimson/zim-utils", "name": "@ztimson/zim-utils",
"version": "0.3.8", "version": "0.4.3",
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js", "description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
"author": "Zak Timson", "author": "Zak Timson",
"license": "MIT", "license": "MIT",
+2 -2
View File
@@ -154,9 +154,9 @@ export class ZimManager {
} }
/** Two-pass fulltext search across every local ZIM, returns enriched results matching catalog/list shape. */ /** Two-pass fulltext search across every local ZIM, returns enriched results matching catalog/list shape. */
async search(terms, limit = 20) { async search(terms, {limit = 20, sources = null} = {}) {
const server = await this.#ensureServer(); const server = await this.#ensureServer();
return server.search(terms, limit); return server.search(terms, {limit, sources});
} }
/** Checks all local ZIMs against the catalog and updates any that are outdated. */ /** Checks all local ZIMs against the catalog and updates any that are outdated. */
+51 -4
View File
@@ -6,6 +6,8 @@ import {decompressPool} from './decompress.js';
const HEADER_SIZE = 80; const HEADER_SIZE = 80;
const NS_CONTENT = 'C'; const NS_CONTENT = 'C';
const NS_METADATA = 'M'; const NS_METADATA = 'M';
const VOCAB_CACHE_LIMIT = 4;
const vocabCache = new Map(); // zimPath -> {mtimeMs, size, words: Set<string>};
async function readAt(fd, pos, length) { async function readAt(fd, pos, length) {
const buf = Buffer.alloc(length); const buf = Buffer.alloc(length);
@@ -33,7 +35,10 @@ async function readMimeTypes(fd, mimeListPos) {
for (;;) { for (;;) {
str += (await readAt(fd, pos, 1024)).toString('binary'); str += (await readAt(fd, pos, 1024)).toString('binary');
const end = str.indexOf('\0\0'); const end = str.indexOf('\0\0');
if (end !== -1) { str = str.slice(0, end + 1); break; } if (end !== -1) {
str = str.slice(0, end + 1);
break;
}
pos += 1024; pos += 1024;
} }
return str.split('\0').filter(Boolean); return str.split('\0').filter(Boolean);
@@ -50,8 +55,15 @@ async function readDirent(fd, offset) {
o += 4; // revision, unused o += 4; // revision, unused
let redirectIndex = null, cluster = null, blob = null; let redirectIndex = null, cluster = null, blob = null;
if (mimetype === 0xffff) { redirectIndex = buf.readUInt32LE(o); o += 4; } if (mimetype === 0xffff) {
else { cluster = buf.readUInt32LE(o); o += 4; blob = buf.readUInt32LE(o); o += 4; } redirectIndex = buf.readUInt32LE(o);
o += 4;
} else {
cluster = buf.readUInt32LE(o);
o += 4;
blob = buf.readUInt32LE(o);
o += 4;
}
const urlEnd = buf.indexOf(0, o); const urlEnd = buf.indexOf(0, o);
if (urlEnd === -1) continue; if (urlEnd === -1) continue;
@@ -73,7 +85,8 @@ async function findByUrl(fd, header, url, namespace) {
const dirKey = dirent.namespace + dirent.url; const dirKey = dirent.namespace + dirent.url;
const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0; const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0;
if (cmp === 0) return dirent; if (cmp === 0) return dirent;
if (cmp < 0) hi = mid - 1; else lo = mid + 1; if (cmp < 0) hi = mid - 1;
else lo = mid + 1;
} }
return null; return null;
} }
@@ -136,3 +149,37 @@ export async function readZimMetadata(zimPath) {
author: creator, publisher, author: creator, publisher,
}; };
} }
/**
* Unique, lowercased words pulled from every article title in a ZIM.
* Only the deduped Set is cached (not the raw title list), and only for the
* last VOCAB_CACHE_LIMIT ZIMs touched (evicted least-recently-used), so memory scales with recent search activity rather than total library size.
* First call per ZIM does a full O(articleCount) directory walk; cached after that until the file's mtime/size changes.
*/
export async function titleVocabulary(zimPath) {
const stat = await fs.promises.stat(zimPath);
const cached = vocabCache.get(zimPath);
if (cached && cached.mtimeMs === stat.mtimeMs && cached.size === stat.size) {
vocabCache.delete(zimPath);
vocabCache.set(zimPath, cached); // bump to most-recently-used
return cached.words;
}
const fd = await fs.promises.open(zimPath, 'r');
let words;
try {
const header = await readHeader(fd);
words = new Set();
for (let i = 0; i < header.articleCount; i++) {
const dirent = await readDirent(fd, await ptr64(fd, header.urlPtrPos, i));
if (dirent.namespace !== NS_CONTENT || dirent.mimetype === 0xffff) continue; // skip redirects/non-content
for (const w of (dirent.title || dirent.url).toLowerCase().split(/\W+/)) if (w) words.add(w);
}
} finally {
await fd.close();
}
vocabCache.set(zimPath, {mtimeMs: stat.mtimeMs, size: stat.size, words});
if (vocabCache.size > VOCAB_CACHE_LIMIT) vocabCache.delete(vocabCache.keys().next().value);
return words;
}
+225
View File
@@ -0,0 +1,225 @@
'use strict';
import {fuzzyMatch} from './utils.js';
const RRF_K = 60;
/** Merges independently-ranked result lists without comparing their raw Xapian scores. */
export function rrfMerge(groups) {
const merged = new Map();
for(const group of groups) {
group.forEach((hit, rank) => {
const contribution = 1 / (RRF_K + rank + 1);
const entry = merged.get(hit.href);
if(entry) entry.score += contribution;
else merged.set(hit.href, {hit, score: contribution});
});
}
return [...merged.values()]
.sort((a, b) => b.score - a.score)
.map(({hit, score}) => ({...hit, rrf: score}));
}
/**
* Calculates field relevance from exact terms, phrase matches, proximity,
* match density and optionally fuzzy similarity.
*/
function fieldScore(text, terms, {fuzzy = false} = {}) {
const value = String(text || '').trim().toLowerCase();
if(!value || !terms.length) {
return {
exact: 0,
coverage: 0,
proximity: 0,
density: 0,
fuzzy: 0,
phrase: 0,
score: 0,
};
}
const words = value.split(/\W+/).filter(Boolean);
const normalizedTerms = terms.map(t => t.toLowerCase());
const phrase = normalizedTerms.join(' ');
const exactTerms = normalizedTerms.filter(t => words.includes(t));
const substringTerms = normalizedTerms.filter(t => value.includes(t));
const coverage = exactTerms.length / normalizedTerms.length;
const substringCoverage = substringTerms.length / normalizedTerms.length;
const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage;
const phraseScore = value.includes(phrase) ? 1 : 0;
let proximity = 0;
if(normalizedTerms.length > 1) {
const positions = [];
for(const term of normalizedTerms) {
const index = words.indexOf(term);
if(index !== -1) positions.push(index);
}
if(positions.length > 1) {
const span = Math.max(...positions) - Math.min(...positions);
proximity = 1 / Math.max(1, span);
}
}
const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0);
const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25));
const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0;
const score =
phraseScore * 1 +
exact * 0.8 +
proximity * 0.35 +
substringCoverage * 0.25 +
density * 0.15 +
fuzzyScore * 0.75;
return {
exact,
coverage,
proximity,
density,
fuzzy: fuzzyScore,
phrase: phraseScore,
score,
};
}
/** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */
export function rerank(hits, termList) {
if(!termList.length) return hits;
return hits.map(hit => {
const title = fieldScore(hit.title, termList, {fuzzy: true});
const summary = fieldScore(hit.summary, termList);
const titleLower = (hit.title || '').toLowerCase();
const summaryLower = (hit.summary?._text || '').toLowerCase();
const phrase = termList.join(' ').toLowerCase();
const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0;
const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0;
const allTitleTerms = termList.every(term =>
titleLower.split(/\W+/).includes(term.toLowerCase())
) ? 1 : 0;
const allSummaryTerms = termList.every(term =>
summaryLower.includes(term.toLowerCase())
) ? 1 : 0;
const finalScore =
hit.rrf +
title.score * 1.25 +
summary.score * 0.35 +
titleExactPhrase * 1.5 +
allTitleTerms * 0.75 +
summaryExactPhrase * 0.2 +
allSummaryTerms * 0.15;
return {
...hit,
finalScore,
ranking: {
rrf: hit.rrf,
title: title.score,
summary: summary.score,
phrase: titleExactPhrase,
coverage: title.coverage,
proximity: title.proximity,
fuzzy: title.fuzzy,
},
};
}).sort((a, b) => b.finalScore - a.finalScore);
}
/** Round-robins results between source ZIMs so one archive cannot dominate the page. */
export function diversify(hits, limit) {
const byBook = new Map();
for(const hit of hits)
(byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit);
for(const list of byBook.values())
list.sort((a, b) => b.finalScore - a.finalScore);
const queues = [...byBook.values()];
const out = [];
for(let i = 0; out.length < limit && queues.some(q => q.length); i++) {
const queue = queues[i % queues.length];
if(queue.length) out.push(queue.shift());
}
return out;
}
function bucketByFirstChar(words) {
const buckets = new Map();
for(const word of words) {
const key = word[0];
(buckets.get(key) ?? buckets.set(key, []).get(key)).push(word);
}
return buckets;
}
/** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */
export function bestFuzzyMatch(term, buckets) {
const lower = term.toLowerCase();
const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34));
const candidates = new Set();
// Check every character so a typo in the first character doesn't eliminate the correct word.
for(const char of lower) {
const bucket = buckets.get(char);
if(bucket) for(const candidate of bucket) candidates.add(candidate);
}
let best = null;
let bestScore = 0;
for(const candidate of candidates) {
if(Math.abs(candidate.length - lower.length) > maxDistance) continue;
const score = fuzzyMatch(candidate, lower).max;
if(score > bestScore) {
bestScore = score;
best = candidate;
}
}
return best;
}
/** Suggests corrected terms from the supplied title vocabulary. */
export function suggestCorrection(termList, vocabulary) {
if(!vocabulary.size) return null;
const buckets = bucketByFirstChar(vocabulary);
let changed = false;
const corrected = termList.map(term => {
if(vocabulary.has(term.toLowerCase())) return term;
const fix = bestFuzzyMatch(term, buckets);
if(fix && fix !== term.toLowerCase()) {
changed = true;
return fix;
}
return term;
});
return changed ? corrected : null;
}
+169 -77
View File
@@ -7,10 +7,12 @@ import fs from 'node:fs';
import path from 'node:path'; import path from 'node:path';
import {fileURLToPath} from 'node:url'; import {fileURLToPath} from 'node:url';
import {fromXml, makeArray} from '@ztimson/utils'; import {fromXml, makeArray} from '@ztimson/utils';
import {titleVocabulary} from './reader.js';
import {diversify, rerank, rrfMerge, suggestCorrection} from './search.js';
const execFileAsync = promisify(execFile); const execFileAsync = promisify(execFile);
const __dirname = path.dirname(fileURLToPath(import.meta.url)); const __dirname = path.dirname(fileURLToPath(import.meta.url));
const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin'); // npm package root/bin - where bin/install.js drops the kiwix-tools binaries const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin');
const READY_TIMEOUT = 10_000; const READY_TIMEOUT = 10_000;
const READY_POLL_INTERVAL = 100; const READY_POLL_INTERVAL = 100;
const WATCH_DEBOUNCE = 300; const WATCH_DEBOUNCE = 300;
@@ -26,9 +28,10 @@ function findFreePort() {
}); });
} }
/** Owns a kiwix-serve process's full lifecycle: library.xml, start/stop/reload, content + search access. function tokenize(terms) {
* Watches its own directory for .zim files appearing/disappearing (from *any* writer - itself, a remote-attached return String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
* ZimManager, a script, whatever) and keeps library.xml + the running kiwix-serve process in sync automatically. */ }
export class KiwixServer { export class KiwixServer {
static #empty = '<?xml version="1.0" encoding="UTF-8" ?>\n<library version="20110515"></library>\n'; static #empty = '<?xml version="1.0" encoding="UTF-8" ?>\n<library version="20110515"></library>\n';
@@ -38,7 +41,7 @@ export class KiwixServer {
#binDir; #binDir;
#libraryPath; #libraryPath;
#child = null; #child = null;
#remote; // baseUrl string if attached to an externally-managed kiwix-serve, else null #remote;
#watcher = null; #watcher = null;
#watchTimer = null; #watchTimer = null;
@@ -46,11 +49,6 @@ export class KiwixServer {
get running() { return !!this.#remote || !!this.#child; } get running() { return !!this.#remote || !!this.#child; }
get baseUrl() { return this.#remote || (this.#child ? `http://${this.#host}:${this.#port}` : null); } get baseUrl() { return this.#remote || (this.#child ? `http://${this.#host}:${this.#port}` : null); }
/** @param {{port?: number, host?: string, binDir?: string, url?: string}} [opts]
* url: attach to an already-running kiwix-serve (e.g. one started elsewhere in your codebase) instead of
* spawning/owning one - start/stop become no-ops, and library.xml is read over HTTP instead of disk.
* Whichever instance *does* own the process is responsible for watching the shared directory; an attached
* instance doesn't need its own watcher; it just reads whatever the owner already reloaded. */
constructor(dir, {port, host = '127.0.0.1', binDir = DEFAULT_BIN_DIR, url} = {}) { constructor(dir, {port, host = '127.0.0.1', binDir = DEFAULT_BIN_DIR, url} = {}) {
this.#dir = dir; this.#dir = dir;
this.#host = host; this.#host = host;
@@ -58,25 +56,29 @@ export class KiwixServer {
this.#binDir = binDir; this.#binDir = binDir;
this.#libraryPath = path.join(dir, 'library.xml'); this.#libraryPath = path.join(dir, 'library.xml');
this.#remote = url ? url.replace(/\/$/, '') : null; this.#remote = url ? url.replace(/\/$/, '') : null;
if (!this.#remote) this.#ensureLocalStore();
if(!this.#remote) this.#ensureLocalStore();
} }
#ensureLocalStore() { #ensureLocalStore() {
fs.mkdirSync(this.#dir, {recursive: true}); fs.mkdirSync(this.#dir, {recursive: true});
if (!fs.existsSync(this.#libraryPath)) fs.writeFileSync(this.#libraryPath, KiwixServer.#empty);
if(!fs.existsSync(this.#libraryPath))
fs.writeFileSync(this.#libraryPath, KiwixServer.#empty);
} }
#assertRunning() { #assertRunning() {
if (!this.running) throw new Error('KiwixServer is not running - call start() first'); if(!this.running) throw new Error('KiwixServer is not running - call start() first');
} }
#bin(name) { #bin(name) {
return path.join(this.#binDir, process.platform === 'win32' ? `${name}.exe` : name); return path.join(this.#binDir, process.platform === 'win32' ? `${name}.exe` : name);
} }
/** Reads library.xml from disk if we own the server, or over HTTP if attached to a remote one. */
async #fetchLibraryXml() { async #fetchLibraryXml() {
if (!this.#remote) return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => ''); if(!this.#remote)
return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => '');
try { try {
const res = await fetch(`${this.#remote}/library.xml`); const res = await fetch(`${this.#remote}/library.xml`);
return res.ok ? await res.text() : ''; return res.ok ? await res.text() : '';
@@ -85,44 +87,55 @@ export class KiwixServer {
} }
} }
/** Rebuilds library.xml from scratch by scanning `dir` for .zim files - no-op if attached to a remote server.
* Rewriting the file bumps its mtime, which kiwix-serve (started with --monitorLibrary) picks up on its own
* and hot-reloads without needing a restart. */
async #rebuildLibrary() { async #rebuildLibrary() {
if (this.#remote) return; if(this.#remote) return;
await fs.promises.rm(this.#libraryPath, {force: true}); await fs.promises.rm(this.#libraryPath, {force: true});
const files = await this.#zimFiles(); const files = await this.#zimFiles();
if (!files.length) return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty);
for (const f of files) await execFileAsync(this.#bin('kiwix-manage'), [this.#libraryPath, 'add', path.join(this.#dir, f)]); if(!files.length)
return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty);
for(const file of files)
await execFileAsync(this.#bin('kiwix-manage'), [
this.#libraryPath,
'add',
path.join(this.#dir, file),
]);
} }
async #waitUntilReady() { async #waitUntilReady() {
const deadline = Date.now() + READY_TIMEOUT; const deadline = Date.now() + READY_TIMEOUT;
while (Date.now() < deadline) {
while(Date.now() < deadline) {
try { try {
await fetch(`http://${this.#host}:${this.#port}/`); await fetch(`http://${this.#host}:${this.#port}/`);
return; return;
} catch {} } catch {}
await new Promise(r => setTimeout(r, READY_POLL_INTERVAL)); await new Promise(r => setTimeout(r, READY_POLL_INTERVAL));
} }
throw new Error('kiwix-serve did not become ready in time'); throw new Error('kiwix-serve did not become ready in time');
} }
async #zimFiles() { async #zimFiles() {
return (await fs.promises.readdir(this.#dir).catch(() => [])).filter(f => f.endsWith('.zim')); return (await fs.promises.readdir(this.#dir).catch(() => []))
.filter(f => f.endsWith('.zim'));
} }
/** Watches `dir` for .zim files being added/removed by anyone (this process, a remote-attached ZimManager,
* a manual copy) and debounces a library rebuild so kiwix-serve's --monitorLibrary picks it up. Only runs
* while we own the process - an attached instance has nothing local to watch on behalf of. */
#watchDir() { #watchDir() {
this.#watcher?.close(); this.#watcher?.close();
this.#watcher = fs.watch(this.#dir, (_event, filename) => { this.#watcher = fs.watch(this.#dir, (_event, filename) => {
if (!filename?.endsWith('.zim')) return; if(!filename?.endsWith('.zim')) return;
clearTimeout(this.#watchTimer); clearTimeout(this.#watchTimer);
this.#watchTimer = setTimeout(() => this.#rebuildLibrary().catch(() => {}), WATCH_DEBOUNCE); this.#watchTimer = setTimeout(() => this.#rebuildLibrary().catch(() => {}), WATCH_DEBOUNCE);
}); });
this.#watcher.on('error', () => {}); // e.g. dir removed out from under us - just stop watching, don't crash the process
this.#watcher.on('error', () => {});
} }
#unwatchDir() { #unwatchDir() {
@@ -132,23 +145,31 @@ export class KiwixServer {
this.#watcher = null; this.#watcher = null;
} }
/** Rebuilds library.xml and starts kiwix-serve. Resolves once the server is responding.
* Starts with --monitorLibrary so the process reloads itself whenever library.xml's mtime changes -
* no restart needed for updates after this. */
async start() { async start() {
if (this.#remote || this.#child) return; if(this.#remote || this.#child) return;
await fs.promises.mkdir(this.#dir, {recursive: true}); await fs.promises.mkdir(this.#dir, {recursive: true});
await this.#rebuildLibrary(); await this.#rebuildLibrary();
this.#port ??= await findFreePort(); this.#port ??= await findFreePort();
this.#child = spawn(this.#bin('kiwix-serve'), this.#child = spawn(this.#bin('kiwix-serve'), [
['--library', '--monitorLibrary', '-i', this.#host, '-p', String(this.#port), this.#libraryPath], '--library',
{stdio: 'ignore'}); '--monitorLibrary',
this.#child.on('exit', () => { this.#child = null; this.#unwatchDir(); }); '-i',
this.#host,
'-p',
String(this.#port),
this.#libraryPath,
], {stdio: 'ignore'});
this.#child.on('exit', () => {
this.#child = null;
this.#unwatchDir();
});
try { try {
await this.#waitUntilReady(); await this.#waitUntilReady();
} catch (e) { } catch(e) {
await this.stop(); await this.stop();
throw e; throw e;
} }
@@ -156,15 +177,18 @@ export class KiwixServer {
this.#watchDir(); this.#watchDir();
} }
/** Gracefully stops kiwix-serve, if we own it. No-op if attached to a remote instance. */
async stop() { async stop() {
this.#unwatchDir(); this.#unwatchDir();
if (this.#remote || !this.#child) return;
if(this.#remote || !this.#child) return;
const child = this.#child; const child = this.#child;
await new Promise(resolve => { await new Promise(resolve => {
child.once('exit', resolve); child.once('exit', resolve);
child.kill('SIGTERM'); child.kill('SIGTERM');
}); });
this.#child = null; this.#child = null;
} }
@@ -173,24 +197,23 @@ export class KiwixServer {
await this.start(); await this.start();
} }
/** Forces an immediate library rebuild rather than waiting for the directory watcher's debounce to fire.
* Not required for correctness (the watcher does this automatically for any writer), just a manual way to
* skip the ~300ms wait when you already know something changed. No-op if attached to a remote instance -
* whoever owns that process already reloads itself. */
async reload() { async reload() {
if (this.#remote || !this.#child) return; if(this.#remote || !this.#child) return;
await this.#rebuildLibrary(); await this.#rebuildLibrary();
} }
/** Local catalog listing - same flat shape as the online catalog (catalog.js), plus a `file` field. */
async list() { async list() {
this.#assertRunning(); this.#assertRunning();
const xml = await this.#fetchLibraryXml(); const xml = await this.#fetchLibraryXml();
if (!xml) return []; if(!xml) return [];
const entries = fromXml(xml); const entries = fromXml(xml);
return makeArray(entries?.library?.book || []).map(e => { return makeArray(entries?.library?.book || []).map(e => {
const tags = e.tags.split(';'); const tags = e.tags.split(';');
const name = e.path.replace('.zim', ''); const name = e.path.replace(/\.zim$/i, '');
return { return {
id: e.id, id: e.id,
title: e.title, title: e.title,
@@ -205,6 +228,7 @@ export class KiwixServer {
publisher: e.publisher, publisher: e.publisher,
articleCount: +e.articleCount || 0, articleCount: +e.articleCount || 0,
sizeMb: +(Number(e.size) / 1024).toFixed(1) || 0, sizeMb: +(Number(e.size) / 1024).toFixed(1) || 0,
file: e.path,
href: name, href: name,
icon: `data:${e.faviconMimetype || 'image/png'};base64,${e.favicon}`, icon: `data:${e.faviconMimetype || 'image/png'};base64,${e.favicon}`,
viewer: `${this.baseUrl}/content/${name}`, viewer: `${this.baseUrl}/content/${name}`,
@@ -212,58 +236,126 @@ export class KiwixServer {
}); });
} }
/** Splits a href ("zim/path/to/page") or a full content/viewer URL into {zim, path}. */
#splitHref(href) { #splitHref(href) {
const clean = href.replace(`${this.baseUrl}/content/`, '').replace(/^\/+/, ''); const clean = href.replace(`${this.baseUrl}/content/`, '').replace(/^\/+/, '');
const [zim, ...rest] = clean.split('/'); const [zim, ...rest] = clean.split('/');
return {zim, path: rest.join('/')}; return {zim, path: rest.join('/')};
} }
/** Builds a kiwix-serve content URL from a href (as returned by list()/search()), or from an already-built content/viewer URL. */
link(href) { link(href) {
this.#assertRunning(); this.#assertRunning();
const {zim, path} = this.#splitHref(href); const {zim, path} = this.#splitHref(href);
return `${this.baseUrl}/content/${zim}${path ? '/' + path : ''}`; return `${this.baseUrl}/content/${zim}${path ? '/' + path : ''}`;
} }
/** Fetches a single asset's raw bytes straight from kiwix-serve. */
async raw(href) { async raw(href) {
const res = await fetch(this.link(href)); const res = await fetch(this.link(href));
if(!res.ok) return null; if(!res.ok) return null;
return {mimetype: res.headers.get('content-type'), data: Buffer.from(await res.arrayBuffer())};
return {
mimetype: res.headers.get('content-type'),
data: Buffer.from(await res.arrayBuffer()),
};
} }
/** Fulltext search across every local ZIM via kiwix-serve's own xapian index */ async #rawSearch(termList, scoped, bookMap, limit) {
async search(terms, limit = 20) { const perBook = await Promise.all(scoped.map(async book => {
const params = new URLSearchParams({
pattern: termList.join(' '),
format: 'xml',
pageLength: String(limit),
'books.name': book.href,
});
const res = await fetch(`${this.baseUrl}/search?${params}`);
if(!res.ok) return [];
const found = fromXml(await res.text())?.rss?.channel?.item || [];
return makeArray(found).map(hit => {
const resultBook = bookMap.get(hit.book?.title) || book;
const prefix = `/content/${resultBook.href}/`;
const page = hit.link.startsWith(prefix)
? hit.link.slice(prefix.length)
: hit.link.replace(/^\/+/, '');
return {
id: resultBook.id,
title: hit.title,
page,
name: resultBook.name,
publisher: resultBook.publisher,
href: `${resultBook.href}/${page}`,
icon: resultBook.icon,
viewer: this.baseUrl + hit.link,
summary: hit.description,
xapianScore: +hit.score || 0,
};
}).filter(Boolean);
}));
return rrfMerge(perBook);
}
/** Fulltext search across local ZIMs with ranking, diversification and spelling correction. */
async search(terms, {limit = 20, sources = null} = {}) {
this.#assertRunning(); this.#assertRunning();
const termList = String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
if (!termList.length) return [];
const params = new URLSearchParams({pattern: termList.join(' '), format: 'xml', pageLength: String(limit)}); const termList = tokenize(terms);
const res = await fetch(`${this.baseUrl}/search?${params}`); if(!termList.length) return {results: [], spellcheck: null};
if (!res.ok) return [];
const found = fromXml(await res.text())?.rss?.channel?.item || [];
const books = await this.list(); const books = await this.list();
const bookMap = new Map(books.map(b => [b.title, b])); const bookMap = new Map(books.map(b => [b.title, b]));
return found.map(hit => { const scoped = sources?.length
const book = bookMap.get(hit.book.title); ? books.filter(b => sources.includes(b.name) || sources.includes(b.href))
if (!book) return null; : books;
const prefix = `/content/${book.href}/`;
const page = hit.link.startsWith(prefix) ? hit.link.slice(prefix.length) : hit.link.replace(/^\/+/, ''); if(!scoped.length) return {results: [], spellcheck: null};
return {
id: book.id, let activeTerms = termList;
title: hit.title, let hits = await this.#rawSearch(activeTerms, scoped, bookMap, limit);
page, let spellcheck = null;
name: book.name,
publisher: book.publisher, if(!hits.length && !this.#remote) {
href: `${book.href}/${page}`, const vocabulary = new Set();
icon: book.icon,
viewer: this.baseUrl + hit.link, for(const book of scoped) {
summary: hit.description, if(!book.file) continue;
score: +hit.score || 0,
}; try {
}).filter(Boolean); for(const word of await titleVocabulary(path.join(this.#dir, book.file)))
vocabulary.add(word);
} catch {
// Unreadable ZIM - skip it, don't fail the whole search.
}
}
const corrected = suggestCorrection(termList, vocabulary);
if(corrected) {
const retry = await this.#rawSearch(corrected, scoped, bookMap, limit);
if(retry.length) {
hits = retry;
activeTerms = corrected;
spellcheck = {
from: termList.join(' '),
to: corrected.join(' '),
};
}
}
}
const ranked = rerank(hits, activeTerms);
return {
results: diversify(ranked, limit),
spellcheck,
};
} }
} }
+12 -15
View File
@@ -1,39 +1,36 @@
export function levenshtein(a, b) { export function levenshtein(a, b) {
const m = a.length, n = b.length; const m = a.length, n = b.length;
if (!m) return n; if(!m) return n;
if (!n) return m; if(!n) return m;
const dp = Array.from({length: m + 1}, (_, i) => [i, ...Array(n).fill(0)]); const dp = Array.from({length: m + 1}, (_, i) => [i, ...Array(n).fill(0)]);
for (let j = 0; j <= n; j++) dp[0][j] = j; for(let j = 0; j <= n; j++) dp[0][j] = j;
for (let i = 1; i <= m; i++) { for(let i = 1; i <= m; i++) {
for (let j = 1; j <= n; j++) { for(let j = 1; j <= n; j++) {
dp[i][j] = a[i - 1] === b[j - 1] dp[i][j] = a[i - 1] === b[j - 1] ? dp[i - 1][j - 1] : 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
? dp[i - 1][j - 1]
: 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
} }
} }
return dp[m][n]; return dp[m][n];
} }
function scoreAgainst(text, term) { function scoreAgainst(text, term) {
if (text.includes(term)) return 1 - (text.length - term.length) / text.length * 0.3; if (!text.length || !term.length) return 0;
if (!text.length || !term.length || text[0] !== term[0]) return 0; if (text === term) return 1;
const dist = levenshtein(text, term); if (text.includes(term)) return 0.8;
const maxAllowed = Math.max(1, Math.ceil(term.length * 0.34)); const maxAllowed = Math.max(1, Math.ceil(term.length * 0.34));
if (Math.abs(text.length - term.length) > maxAllowed) return 0;
const dist = levenshtein(text, term);
if (dist > maxAllowed) return 0; if (dist > maxAllowed) return 0;
return 1 - dist / Math.max(text.length, term.length); return 1 - dist / Math.max(text.length, term.length);
} }
/** Compares `target` against one or more search terms; returns avg/max/per-term similarity. */
export function fuzzyMatch(target, ...terms) { export function fuzzyMatch(target, ...terms) {
if (!terms.length) throw new Error('Requires at least 1 term to compare'); if (!terms.length) throw new Error('Requires at least 1 term to compare');
const lowerTarget = String(target).toLowerCase(); const lowerTarget = String(target).toLowerCase();
const words = lowerTarget.split(/\W+/).filter(Boolean); const words = lowerTarget.split(/\W+/).filter(Boolean);
const similarities = terms.map(term => { const similarities = terms.map(term => {
const t = term.toLowerCase(); const t = String(term).toLowerCase();
return Math.max(scoreAgainst(lowerTarget, t), ...words.map(w => scoreAgainst(w, t))); return Math.max(scoreAgainst(lowerTarget, t), ...words.map(w => scoreAgainst(w, t)));
}); });
return { return {
avg: similarities.reduce((acc, s) => acc + s, 0) / similarities.length, avg: similarities.reduce((acc, s) => acc + s, 0) / similarities.length,
max: Math.max(...similarities), max: Math.max(...similarities),