Compare commits

9 Commits
Author SHA1 Message Date
ztimson dc24f48a36 Better catalog search
Publish Library / Build NPM Project (push) Successful in 3m6s
Publish Library / Tag Version (push) Canceled after 9s
2026-09-20 23:27:16 -04:00
ztimson 812e72a8cd Better catalog search
Publish Library / Build NPM Project (push) Successful in 4m2s
Publish Library / Tag Version (push) Failing after 19s
2026-09-20 23:26:57 -04:00
ztimson a45d7bad8c Search fix
Publish Library / Build NPM Project (push) Successful in 4m13s
Publish Library / Tag Version (push) Successful in 16s
2026-09-20 20:57:46 -04:00
ztimson 0a7e87f6b7 Patched search results
Publish Library / Build NPM Project (push) Successful in 3m16s
Publish Library / Tag Version (push) Successful in 19s
2026-09-20 20:29:28 -04:00
ztimson e12c3e3cc6 Fixed typo
Publish Library / Build NPM Project (push) Successful in 4m39s
Publish Library / Tag Version (push) Successful in 8s
2026-09-20 19:22:10 -04:00
ztimson 8d4258b951 Better search results
Publish Library / Build NPM Project (push) Successful in 4m47s
Publish Library / Tag Version (push) Successful in 9s
2026-09-20 19:05:25 -04:00
ztimson 15f15e06f1 Merge branch 'master' of git.zakscode.com:ztimson/zim-utils into fix/install-kwix-arm64
Publish Library / Build NPM Project (push) Successful in 4m52s
Publish Library / Tag Version (push) Successful in 14s
2026-09-20 13:04:21 -04:00
ztimson 997bd6064d Auto library reloading
Publish Library / Build NPM Project (push) Successful in 5m45s
Publish Library / Tag Version (push) Successful in 26s
2026-09-20 12:54:56 -04:00
ztimson 4b86fe8051 Removed outdated zims automatically
Publish Library / Build NPM Project (push) Successful in 5m35s
Publish Library / Tag Version (push) Successful in 32s
2026-09-20 10:55:02 -04:00
7 changed files with 621 additions and 118 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "@ztimson/zim-utils", "name": "@ztimson/zim-utils",
"version": "0.3.6", "version": "0.4.4",
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js", "description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
"author": "Zak Timson", "author": "Zak Timson",
"license": "MIT", "license": "MIT",
+101 -21
View File
@@ -2,19 +2,29 @@ import {fuzzyMatch} from './utils.js';
import {decodeHtml} from '@ztimson/utils'; import {decodeHtml} from '@ztimson/utils';
export const CATALOG_URL = 'https://library.kiwix.org'; export const CATALOG_URL = 'https://library.kiwix.org';
export const VIEWER_URL = 'https://browse.library.kiwix.org';
const PAGE_SIZE = 100; const PAGE_SIZE = 100;
export function parseEntries(xml, catalog = CATALOG_URL) { export function parseEntries(xml, catalog = CATALOG_URL) {
const blocks = xml.match(/<entry>[\s\S]*?<\/entry>/g) || []; const blocks = xml.match(/<entry>[\s\S]*?<\/entry>/g) || [];
return blocks.map(b => { return blocks.map(b => {
const grab = re => (b.match(re) || [])[1] || ''; const grab = re => (b.match(re) || [])[1] || '';
const linkMatch = b.match(/<link[^>]*type=["']application\/x-zim[^"']*["'][^>]*href=["']([^"']+)["']/); const linkMatch = b.match(/<link[^>]*type=["']application\/x-zim[^"']*["'][^>]*href=["']([^"']+)["']/);
const iconMatch = b.match(/<link[^>]*rel=["']http:\/\/opds-spec\.org\/image\/thumbnail["'][^>]*href=["']([^"']+)["']/) const iconMatch = b.match(/<link[^>]*rel=["']http:\/\/opds-spec\.org\/image\/thumbnail["'][^>]*href=["']([^"']+)["']/)
|| b.match(/<link[^>]*type=["']image\/[^"']*["'][^>]*href=["']([^"']+)["']/); || b.match(/<link[^>]*type=["']image\/[^"']*["'][^>]*href=["']([^"']+)["']/);
let tags = grab(/<tags>([^<]*)<\/tags>/); let tags = grab(/<tags>([^<]*)<\/tags>/);
if (tags) tags = tags.split(';'); if (tags) tags = tags.split(';');
const name = grab(/<name>([^<]*)<\/name>/); const name = grab(/<name>([^<]*)<\/name>/);
const href = linkMatch ? linkMatch[1] : null;
const filename = href
? decodeURIComponent(new URL(href).pathname.split('/').pop()).replace(/\.meta4$/i, '')
: name;
const viewerName = filename.replace(/\.zim$/i, '');
return { return {
id: grab(/<id>([^<]*)<\/id>/), id: grab(/<id>([^<]*)<\/id>/),
title: decodeHtml(grab(/<title>([^<]*)<\/title>/)), title: decodeHtml(grab(/<title>([^<]*)<\/title>/)),
@@ -22,25 +32,52 @@ export function parseEntries(xml, catalog = CATALOG_URL) {
summary: decodeHtml(grab(/<summary>([^<]*)<\/summary>/)), summary: decodeHtml(grab(/<summary>([^<]*)<\/summary>/)),
language: grab(/<language>([^<]*)<\/language>/), language: grab(/<language>([^<]*)<\/language>/),
name, name,
flavour: grab(/<flavour>([^<]*)<\/flavour>/),
category: grab(/<category>([^<]*)<\/category>/), category: grab(/<category>([^<]*)<\/category>/),
tags, tags,
mediaCount: Number(grab(/<mediaCount>([^<]*)<\/mediaCount>/)) || 0, mediaCount: Number(grab(/<mediaCount>([^<]*)<\/mediaCount>/)) || 0,
author: grab(/<author>\s*<name>([^<]*)<\/name>\s*<\/author>/m), author: grab(/<author>\s*<name>([^<]*)<\/name>\s*<\/author>/m),
publisher: grab(/<publisher>\s*<name>([^<]*)<\/name>\s*<\/publisher>/m), publisher: grab(/<publisher>\s*<name>([^<]*)<\/name>\s*<\/publisher>/m),
articleCount: Number(grab(/<articleCount>([^<]*)<\/articleCount>/)) || 0, articleCount: Number(grab(/<articleCount>([^<]*)<\/articleCount>/)) || 0,
sizeMb: linkMatch ? +(Number((b.match(/length=["'](\d+)["']/) || [])[1] || 0) / 1024 / 1024).toFixed(1) : '?', sizeMb: href
href: linkMatch ? linkMatch[1] : null, ? +(Number((b.match(/length=["'](\d+)["']/) || [])[1] || 0) / 1024 / 1024).toFixed(1)
: '?',
href,
icon: iconMatch ? new URL(iconMatch[1], catalog).href : null, icon: iconMatch ? new URL(iconMatch[1], catalog).href : null,
viewer: name ? new URL(`viewer#${name}`, catalog).href : null, viewer: viewerName ? `${VIEWER_URL}/viewer#${viewerName}` : null,
}; };
}); });
} }
async function fetchEntries(term, lang, url = CATALOG_URL) { async function fetchEntries(term, lang, url = CATALOG_URL) {
const params = new URLSearchParams({q: term, count: String(PAGE_SIZE), lang: lang || 'eng'}); const entries = [];
let start = 0;
let total = Infinity;
while (start < total) {
const params = new URLSearchParams({
q: term,
count: String(PAGE_SIZE),
start: String(start),
lang: lang || 'eng',
});
const res = await fetch(`${url}/catalog/v2/entries?${params}`); const res = await fetch(`${url}/catalog/v2/entries?${params}`);
if (!res.ok) throw new Error(`${res.status} ${res.statusText}`); if (!res.ok) throw new Error(`${res.status} ${res.statusText}`);
return parseEntries(await res.text());
const xml = await res.text();
const page = parseEntries(xml, url);
const totalResults = Number((xml.match(/<totalResults>(\d+)<\/totalResults>/) || [])[1]);
entries.push(...page);
if (Number.isFinite(totalResults)) total = totalResults;
if (!page.length || start + page.length >= total) break;
start += page.length;
}
return entries;
} }
/** Looks up a single catalog entry by exact `name` (used to check for available updates). */ /** Looks up a single catalog entry by exact `name` (used to check for available updates). */
@@ -48,30 +85,73 @@ export async function zimCatalogInfo(name, url = CATALOG_URL) {
const params = new URLSearchParams({name, count: '5'}); const params = new URLSearchParams({name, count: '5'});
const res = await fetch(`${url}/catalog/v2/entries?${params}`); const res = await fetch(`${url}/catalog/v2/entries?${params}`);
if (!res.ok) return null; if (!res.ok) return null;
return parseEntries(await res.text()).find(e => e.name === name) || null;
return parseEntries(await res.text(), url).find(e => e.name === name) || null;
} }
/** Searches the Kiwix catalog for ZIMs matching `terms` within `category`, ranked by term coverage then fuzzy similarity. */ /** Searches the Kiwix catalog for ZIMs matching `terms`, ranked by relevance. */
export async function zimCatalog(terms, opts = {lang: 'eng', count: 20, url: CATALOG_URL}) { export async function zimCatalog(terms, opts = {lang: 'eng', count: 20, url: CATALOG_URL}) {
opts = Object.assign({lang: 'eng', count: 20, url: CATALOG_URL}, opts) opts = Object.assign({lang: 'eng', count: 20, url: CATALOG_URL}, opts);
const termList = [...String(terms).split(',')].filter(Boolean).map(t => t.trim().toLowerCase());
const results = await Promise.allSettled(termList.map(t => fetchEntries(t, opts.lang, opts.url))); const termList = [...String(terms).split(',')]
const byName = new Map(); .filter(Boolean)
.map(t => t.trim().toLowerCase());
const results = await Promise.allSettled(
termList.map(t => fetchEntries(t, opts.lang, opts.url))
);
const byZim = new Map();
results.forEach((r, i) => { results.forEach((r, i) => {
if (r.status !== 'fulfilled') return; if (r.status !== 'fulfilled') return;
const term = termList[i]; const term = termList[i];
for (const entry of r.value) { for (const entry of r.value) {
if (!entry.name) continue; const zimId = entry.href || entry.id || entry.name;
if (!byName.has(entry.name)) byName.set(entry.name, {entry, hitTerms: new Set()}); if (!zimId) continue;
byName.get(entry.name).hitTerms.add(term);
if (!byZim.has(zimId)) {
byZim.set(zimId, {entry, hitTerms: new Set()});
}
byZim.get(zimId).hitTerms.add(term);
} }
}); });
if (!byName.size) return [];
return [...byName.values()].map(({entry, hitTerms}) => { if (!byZim.size) return [];
const text = `${entry.title} ${entry.summary}`.trim();
return {entry, hits: hitTerms.size, fuzzy: fuzzyMatch(text, ...termList).max}; return [...byZim.values()]
}).toSorted((a, b) => .map(({entry, hitTerms}) => {
b.hits - a.hits || b.fuzzy - a.fuzzy || b.entry.articleCount - a.entry.articleCount const title = entry.title.toLowerCase();
).slice(0, opts.count).map(r => r.entry); const summary = entry.summary.toLowerCase();
const titleWords = title.split(/\W+/).filter(Boolean);
const exactTitle = termList.some(term => title === term);
const exactWord = termList.some(term => titleWords.includes(term));
const titleMatch = termList.some(term => title.includes(term));
const titleFuzzy = fuzzyMatch(title, ...termList).max;
const summaryFuzzy = fuzzyMatch(summary, ...termList).max;
return {
entry,
hits: hitTerms.size,
exactTitle,
exactWord,
titleMatch,
titleFuzzy,
summaryFuzzy,
};
})
.toSorted((a, b) =>
Number(b.exactTitle) - Number(a.exactTitle)
|| Number(b.exactWord) - Number(a.exactWord)
|| Number(b.titleMatch) - Number(a.titleMatch)
|| b.titleFuzzy - a.titleFuzzy
|| b.summaryFuzzy - a.summaryFuzzy
|| b.hits - a.hits
|| new Date(b.entry.updated) - new Date(a.entry.updated)
)
.map(r => r.entry);
} }
+34 -10
View File
@@ -5,8 +5,13 @@ import {Readable} from 'node:stream';
import {KiwixServer} from './server.js'; import {KiwixServer} from './server.js';
import {zimCatalog, zimCatalogInfo, CATALOG_URL} from './catalog.js'; import {zimCatalog, zimCatalogInfo, CATALOG_URL} from './catalog.js';
const VISIBLE_TIMEOUT = 5_000;
const VISIBLE_POLL_INTERVAL = 200;
/** Manages a local directory of ZIM archives: catalog search, downloads, update checks, deletion. /** Manages a local directory of ZIM archives: catalog search, downloads, update checks, deletion.
* Reuses (or owns & lazily starts) a KiwixServer for local listing/search, so `list()` and `catalog()` return the same shape. */ * Reuses (or owns & lazily starts) a KiwixServer for local listing/search, so `list()` and `catalog()` return the same shape.
* Doesn't drive library.xml/kiwix-serve reloads itself - the owning KiwixServer watches its directory and
* reloads automatically whenever .zim files change, no matter which ZimManager (local or remote-attached) wrote them. */
export class ZimManager { export class ZimManager {
#catalogUrl; #catalogUrl;
#dir; #dir;
@@ -24,6 +29,10 @@ export class ZimManager {
this.#server = server ?? new KiwixServer(dir, {port, host, binDir, url}); this.#server = server ?? new KiwixServer(dir, {port, host, binDir, url});
} }
#stripDate(filename) {
return filename.replace(/\.zim$/i, '').replace(/_\d{4}-\d{2}(?:_\d+)?$/, '');
}
/** The KiwixServer backing this manager - reuse it directly for content/search access, or pass into another ZimManager. */ /** The KiwixServer backing this manager - reuse it directly for content/search access, or pass into another ZimManager. */
get server() { return this.#server; } get server() { return this.#server; }
@@ -52,6 +61,20 @@ export class ZimManager {
return {res, url: m[1]}; return {res, url: m[1]};
} }
/** Polls list() until `filename` shows up (or disappears, if `expect: false`), so callers get an accurate
* status back rather than one that's ahead of what the server has actually picked up yet. The owning
* KiwixServer's directory watcher does the real reload work in the background; this just waits for it. */
async #waitUntilVisible(filename, {expect = true, timeout = VISIBLE_TIMEOUT} = {}) {
const deadline = Date.now() + timeout;
while (Date.now() < deadline) {
const local = await this.list();
const present = local.some(l => l.file === filename);
if (present === expect) return true;
await new Promise(r => setTimeout(r, VISIBLE_POLL_INTERVAL));
}
return false;
}
async #update(name, catalogEntry, localMatch, force) { async #update(name, catalogEntry, localMatch, force) {
const remoteDate = catalogEntry.updated ? new Date(catalogEntry.updated) : null; const remoteDate = catalogEntry.updated ? new Date(catalogEntry.updated) : null;
const localDate = localMatch?.updated ?? null; const localDate = localMatch?.updated ?? null;
@@ -63,8 +86,8 @@ export class ZimManager {
const destPath = path.join(this.#dir, filename); const destPath = path.join(this.#dir, filename);
await this.#download(catalogEntry.href, destPath); await this.#download(catalogEntry.href, destPath);
if (localMatch && localMatch.file !== filename) await fs.promises.rm(path.join(this.#dir, localMatch.file), {force: true}); if (localMatch && localMatch.file !== filename) await fs.promises.rm(path.join(this.#dir, localMatch.file), {force: true});
const server = await this.#ensureServer(); await this.#ensureServer();
await server.reload(); await this.#waitUntilVisible(filename, {expect: true});
return {name, status: 'updated', file: filename}; return {name, status: 'updated', file: filename};
} }
@@ -83,8 +106,8 @@ export class ZimManager {
if (!match) throw new Error(`ZIM not found locally: ${nameOrFile}`); if (!match) throw new Error(`ZIM not found locally: ${nameOrFile}`);
const file = match.href + (match.href.endsWith('.zim') ? '' : '.zim'); const file = match.href + (match.href.endsWith('.zim') ? '' : '.zim');
await fs.promises.rm(path.join(this.#dir, file), {force: true}); await fs.promises.rm(path.join(this.#dir, file), {force: true});
const server = await this.#ensureServer(); await this.#ensureServer();
await server.reload(); await this.#waitUntilVisible(file, {expect: false});
return {file: match.file, status: 'deleted'}; return {file: match.file, status: 'deleted'};
} }
@@ -93,11 +116,12 @@ export class ZimManager {
const {url: finalUrl} = await this.#resolveUrl(href); const {url: finalUrl} = await this.#resolveUrl(href);
const filename = path.basename(new URL(finalUrl).pathname).replace(/\.meta4$/i, ''); const filename = path.basename(new URL(finalUrl).pathname).replace(/\.meta4$/i, '');
await fs.promises.mkdir(this.#dir, {recursive: true}); await fs.promises.mkdir(this.#dir, {recursive: true});
const name = filename.replace(/\.zim$/i, '').replace(/_\d{4}-\d{2}(?:_\d+)?$/, '');
const local = await this.list(); const local = await this.list();
const exact = local.find(l => l.file === filename);
if (exact && !force) return {name: exact.name ?? this.#stripDate(filename), status: 'skipped', reason: 'already downloaded', file: filename};
const name = this.#stripDate(filename);
const localMatch = local.find(l => l.name === name) ?? null; const localMatch = local.find(l => l.name === name) ?? null;
const catalogEntry = await zimCatalogInfo(name, this.#catalogUrl) || {name, updated: null, href: href}; const catalogEntry = await zimCatalogInfo(name, this.#catalogUrl) || {name, updated: null, href};
return this.#update(name, catalogEntry, localMatch, force); return this.#update(name, catalogEntry, localMatch, force);
} }
@@ -130,9 +154,9 @@ export class ZimManager {
} }
/** Two-pass fulltext search across every local ZIM, returns enriched results matching catalog/list shape. */ /** Two-pass fulltext search across every local ZIM, returns enriched results matching catalog/list shape. */
async search(terms, limit = 20) { async search(terms, {limit = 20, sources = null} = {}) {
const server = await this.#ensureServer(); const server = await this.#ensureServer();
return server.search(terms, limit); return server.search(terms, {limit, sources});
} }
/** Checks all local ZIMs against the catalog and updates any that are outdated. */ /** Checks all local ZIMs against the catalog and updates any that are outdated. */
+51 -4
View File
@@ -6,6 +6,8 @@ import {decompressPool} from './decompress.js';
const HEADER_SIZE = 80; const HEADER_SIZE = 80;
const NS_CONTENT = 'C'; const NS_CONTENT = 'C';
const NS_METADATA = 'M'; const NS_METADATA = 'M';
const VOCAB_CACHE_LIMIT = 4;
const vocabCache = new Map(); // zimPath -> {mtimeMs, size, words: Set<string>};
async function readAt(fd, pos, length) { async function readAt(fd, pos, length) {
const buf = Buffer.alloc(length); const buf = Buffer.alloc(length);
@@ -33,7 +35,10 @@ async function readMimeTypes(fd, mimeListPos) {
for (;;) { for (;;) {
str += (await readAt(fd, pos, 1024)).toString('binary'); str += (await readAt(fd, pos, 1024)).toString('binary');
const end = str.indexOf('\0\0'); const end = str.indexOf('\0\0');
if (end !== -1) { str = str.slice(0, end + 1); break; } if (end !== -1) {
str = str.slice(0, end + 1);
break;
}
pos += 1024; pos += 1024;
} }
return str.split('\0').filter(Boolean); return str.split('\0').filter(Boolean);
@@ -50,8 +55,15 @@ async function readDirent(fd, offset) {
o += 4; // revision, unused o += 4; // revision, unused
let redirectIndex = null, cluster = null, blob = null; let redirectIndex = null, cluster = null, blob = null;
if (mimetype === 0xffff) { redirectIndex = buf.readUInt32LE(o); o += 4; } if (mimetype === 0xffff) {
else { cluster = buf.readUInt32LE(o); o += 4; blob = buf.readUInt32LE(o); o += 4; } redirectIndex = buf.readUInt32LE(o);
o += 4;
} else {
cluster = buf.readUInt32LE(o);
o += 4;
blob = buf.readUInt32LE(o);
o += 4;
}
const urlEnd = buf.indexOf(0, o); const urlEnd = buf.indexOf(0, o);
if (urlEnd === -1) continue; if (urlEnd === -1) continue;
@@ -73,7 +85,8 @@ async function findByUrl(fd, header, url, namespace) {
const dirKey = dirent.namespace + dirent.url; const dirKey = dirent.namespace + dirent.url;
const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0; const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0;
if (cmp === 0) return dirent; if (cmp === 0) return dirent;
if (cmp < 0) hi = mid - 1; else lo = mid + 1; if (cmp < 0) hi = mid - 1;
else lo = mid + 1;
} }
return null; return null;
} }
@@ -136,3 +149,37 @@ export async function readZimMetadata(zimPath) {
author: creator, publisher, author: creator, publisher,
}; };
} }
/**
* Unique, lowercased words pulled from every article title in a ZIM.
* Only the deduped Set is cached (not the raw title list), and only for the
* last VOCAB_CACHE_LIMIT ZIMs touched (evicted least-recently-used), so memory scales with recent search activity rather than total library size.
* First call per ZIM does a full O(articleCount) directory walk; cached after that until the file's mtime/size changes.
*/
export async function titleVocabulary(zimPath) {
const stat = await fs.promises.stat(zimPath);
const cached = vocabCache.get(zimPath);
if (cached && cached.mtimeMs === stat.mtimeMs && cached.size === stat.size) {
vocabCache.delete(zimPath);
vocabCache.set(zimPath, cached); // bump to most-recently-used
return cached.words;
}
const fd = await fs.promises.open(zimPath, 'r');
let words;
try {
const header = await readHeader(fd);
words = new Set();
for (let i = 0; i < header.articleCount; i++) {
const dirent = await readDirent(fd, await ptr64(fd, header.urlPtrPos, i));
if (dirent.namespace !== NS_CONTENT || dirent.mimetype === 0xffff) continue; // skip redirects/non-content
for (const w of (dirent.title || dirent.url).toLowerCase().split(/\W+/)) if (w) words.add(w);
}
} finally {
await fd.close();
}
vocabCache.set(zimPath, {mtimeMs: stat.mtimeMs, size: stat.size, words});
if (vocabCache.size > VOCAB_CACHE_LIMIT) vocabCache.delete(vocabCache.keys().next().value);
return words;
}
+225
View File
@@ -0,0 +1,225 @@
'use strict';
import {fuzzyMatch} from './utils.js';
const RRF_K = 60;
/** Merges independently-ranked result lists without comparing their raw Xapian scores. */
export function rrfMerge(groups) {
const merged = new Map();
for(const group of groups) {
group.forEach((hit, rank) => {
const contribution = 1 / (RRF_K + rank + 1);
const entry = merged.get(hit.href);
if(entry) entry.score += contribution;
else merged.set(hit.href, {hit, score: contribution});
});
}
return [...merged.values()]
.sort((a, b) => b.score - a.score)
.map(({hit, score}) => ({...hit, rrf: score}));
}
/**
* Calculates field relevance from exact terms, phrase matches, proximity,
* match density and optionally fuzzy similarity.
*/
function fieldScore(text, terms, {fuzzy = false} = {}) {
const value = String(text || '').trim().toLowerCase();
if(!value || !terms.length) {
return {
exact: 0,
coverage: 0,
proximity: 0,
density: 0,
fuzzy: 0,
phrase: 0,
score: 0,
};
}
const words = value.split(/\W+/).filter(Boolean);
const normalizedTerms = terms.map(t => t.toLowerCase());
const phrase = normalizedTerms.join(' ');
const exactTerms = normalizedTerms.filter(t => words.includes(t));
const substringTerms = normalizedTerms.filter(t => value.includes(t));
const coverage = exactTerms.length / normalizedTerms.length;
const substringCoverage = substringTerms.length / normalizedTerms.length;
const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage;
const phraseScore = value.includes(phrase) ? 1 : 0;
let proximity = 0;
if(normalizedTerms.length > 1) {
const positions = [];
for(const term of normalizedTerms) {
const index = words.indexOf(term);
if(index !== -1) positions.push(index);
}
if(positions.length > 1) {
const span = Math.max(...positions) - Math.min(...positions);
proximity = 1 / Math.max(1, span);
}
}
const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0);
const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25));
const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0;
const score =
phraseScore * 1 +
exact * 0.8 +
proximity * 0.35 +
substringCoverage * 0.25 +
density * 0.15 +
fuzzyScore * 0.75;
return {
exact,
coverage,
proximity,
density,
fuzzy: fuzzyScore,
phrase: phraseScore,
score,
};
}
/** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */
export function rerank(hits, termList) {
if(!termList.length) return hits;
return hits.map(hit => {
const title = fieldScore(hit.title, termList, {fuzzy: true});
const summary = fieldScore(hit.summary, termList);
const titleLower = (hit.title || '').toLowerCase();
const summaryLower = (hit.summary?._text || '').toLowerCase();
const phrase = termList.join(' ').toLowerCase();
const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0;
const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0;
const allTitleTerms = termList.every(term =>
titleLower.split(/\W+/).includes(term.toLowerCase())
) ? 1 : 0;
const allSummaryTerms = termList.every(term =>
summaryLower.includes(term.toLowerCase())
) ? 1 : 0;
const finalScore =
hit.rrf +
title.score * 1.25 +
summary.score * 0.35 +
titleExactPhrase * 1.5 +
allTitleTerms * 0.75 +
summaryExactPhrase * 0.2 +
allSummaryTerms * 0.15;
return {
...hit,
finalScore,
ranking: {
rrf: hit.rrf,
title: title.score,
summary: summary.score,
phrase: titleExactPhrase,
coverage: title.coverage,
proximity: title.proximity,
fuzzy: title.fuzzy,
},
};
}).sort((a, b) => b.finalScore - a.finalScore);
}
/** Round-robins results between source ZIMs so one archive cannot dominate the page. */
export function diversify(hits, limit) {
const byBook = new Map();
for(const hit of hits)
(byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit);
for(const list of byBook.values())
list.sort((a, b) => b.finalScore - a.finalScore);
const queues = [...byBook.values()];
const out = [];
for(let i = 0; out.length < limit && queues.some(q => q.length); i++) {
const queue = queues[i % queues.length];
if(queue.length) out.push(queue.shift());
}
return out;
}
function bucketByFirstChar(words) {
const buckets = new Map();
for(const word of words) {
const key = word[0];
(buckets.get(key) ?? buckets.set(key, []).get(key)).push(word);
}
return buckets;
}
/** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */
export function bestFuzzyMatch(term, buckets) {
const lower = term.toLowerCase();
const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34));
const candidates = new Set();
// Check every character so a typo in the first character doesn't eliminate the correct word.
for(const char of lower) {
const bucket = buckets.get(char);
if(bucket) for(const candidate of bucket) candidates.add(candidate);
}
let best = null;
let bestScore = 0;
for(const candidate of candidates) {
if(Math.abs(candidate.length - lower.length) > maxDistance) continue;
const score = fuzzyMatch(candidate, lower).max;
if(score > bestScore) {
bestScore = score;
best = candidate;
}
}
return best;
}
/** Suggests corrected terms from the supplied title vocabulary. */
export function suggestCorrection(termList, vocabulary) {
if(!vocabulary.size) return null;
const buckets = bucketByFirstChar(vocabulary);
let changed = false;
const corrected = termList.map(term => {
if(vocabulary.has(term.toLowerCase())) return term;
const fix = bestFuzzyMatch(term, buckets);
if(fix && fix !== term.toLowerCase()) {
changed = true;
return fix;
}
return term;
});
return changed ? corrected : null;
}
+180 -50
View File
@@ -6,13 +6,16 @@ import net from 'node:net';
import fs from 'node:fs'; import fs from 'node:fs';
import path from 'node:path'; import path from 'node:path';
import {fileURLToPath} from 'node:url'; import {fileURLToPath} from 'node:url';
import {fromXml} from '@ztimson/utils'; import {fromXml, makeArray} from '@ztimson/utils';
import {titleVocabulary} from './reader.js';
import {diversify, rerank, rrfMerge, suggestCorrection} from './search.js';
const execFileAsync = promisify(execFile); const execFileAsync = promisify(execFile);
const __dirname = path.dirname(fileURLToPath(import.meta.url)); const __dirname = path.dirname(fileURLToPath(import.meta.url));
const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin'); // npm package root/bin - where bin/install.js drops the kiwix-tools binaries const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin');
const READY_TIMEOUT = 10_000; const READY_TIMEOUT = 10_000;
const READY_POLL_INTERVAL = 100; const READY_POLL_INTERVAL = 100;
const WATCH_DEBOUNCE = 300;
function findFreePort() { function findFreePort() {
return new Promise((resolve, reject) => { return new Promise((resolve, reject) => {
@@ -25,7 +28,10 @@ function findFreePort() {
}); });
} }
/** Owns a kiwix-serve process's full lifecycle: library.xml, start/stop/reload, content + search access. */ function tokenize(terms) {
return String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
}
export class KiwixServer { export class KiwixServer {
static #empty = '<?xml version="1.0" encoding="UTF-8" ?>\n<library version="20110515"></library>\n'; static #empty = '<?xml version="1.0" encoding="UTF-8" ?>\n<library version="20110515"></library>\n';
@@ -35,15 +41,14 @@ export class KiwixServer {
#binDir; #binDir;
#libraryPath; #libraryPath;
#child = null; #child = null;
#remote; // baseUrl string if attached to an externally-managed kiwix-serve, else null #remote;
#watcher = null;
#watchTimer = null;
get port() { return this.#port; } get port() { return this.#port; }
get running() { return !!this.#remote || !!this.#child; } get running() { return !!this.#remote || !!this.#child; }
get baseUrl() { return this.#remote || (this.#child ? `http://${this.#host}:${this.#port}` : null); } get baseUrl() { return this.#remote || (this.#child ? `http://${this.#host}:${this.#port}` : null); }
/** @param {{port?: number, host?: string, binDir?: string, url?: string}} [opts]
* url: attach to an already-running kiwix-serve (e.g. one started elsewhere in your codebase) instead of
* spawning/owning one - start/stop/reload become no-ops, and library.xml is read over HTTP instead of disk. */
constructor(dir, {port, host = '127.0.0.1', binDir = DEFAULT_BIN_DIR, url} = {}) { constructor(dir, {port, host = '127.0.0.1', binDir = DEFAULT_BIN_DIR, url} = {}) {
this.#dir = dir; this.#dir = dir;
this.#host = host; this.#host = host;
@@ -51,12 +56,15 @@ export class KiwixServer {
this.#binDir = binDir; this.#binDir = binDir;
this.#libraryPath = path.join(dir, 'library.xml'); this.#libraryPath = path.join(dir, 'library.xml');
this.#remote = url ? url.replace(/\/$/, '') : null; this.#remote = url ? url.replace(/\/$/, '') : null;
if(!this.#remote) this.#ensureLocalStore(); if(!this.#remote) this.#ensureLocalStore();
} }
#ensureLocalStore() { #ensureLocalStore() {
fs.mkdirSync(this.#dir, {recursive: true}); fs.mkdirSync(this.#dir, {recursive: true});
if (!fs.existsSync(this.#libraryPath)) fs.writeFileSync(this.#libraryPath, KiwixServer.#empty);
if(!fs.existsSync(this.#libraryPath))
fs.writeFileSync(this.#libraryPath, KiwixServer.#empty);
} }
#assertRunning() { #assertRunning() {
@@ -67,9 +75,10 @@ export class KiwixServer {
return path.join(this.#binDir, process.platform === 'win32' ? `${name}.exe` : name); return path.join(this.#binDir, process.platform === 'win32' ? `${name}.exe` : name);
} }
/** Reads library.xml from disk if we own the server, or over HTTP if attached to a remote one. */
async #fetchLibraryXml() { async #fetchLibraryXml() {
if (!this.#remote) return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => ''); if(!this.#remote)
return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => '');
try { try {
const res = await fetch(`${this.#remote}/library.xml`); const res = await fetch(`${this.#remote}/library.xml`);
return res.ok ? await res.text() : ''; return res.ok ? await res.text() : '';
@@ -78,40 +87,85 @@ export class KiwixServer {
} }
} }
/** Rebuilds library.xml from scratch by scanning `dir` for .zim files - no-op if attached to a remote server. */
async #rebuildLibrary() { async #rebuildLibrary() {
if(this.#remote) return; if(this.#remote) return;
await fs.promises.rm(this.#libraryPath, {force: true}); await fs.promises.rm(this.#libraryPath, {force: true});
const files = await this.#zimFiles(); const files = await this.#zimFiles();
if (!files.length) return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty);
for (const f of files) await execFileAsync(this.#bin('kiwix-manage'), [this.#libraryPath, 'add', path.join(this.#dir, f)]); if(!files.length)
return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty);
for(const file of files)
await execFileAsync(this.#bin('kiwix-manage'), [
this.#libraryPath,
'add',
path.join(this.#dir, file),
]);
} }
async #waitUntilReady() { async #waitUntilReady() {
const deadline = Date.now() + READY_TIMEOUT; const deadline = Date.now() + READY_TIMEOUT;
while(Date.now() < deadline) { while(Date.now() < deadline) {
try { try {
await fetch(`http://${this.#host}:${this.#port}/`); await fetch(`http://${this.#host}:${this.#port}/`);
return; return;
} catch {} } catch {}
await new Promise(r => setTimeout(r, READY_POLL_INTERVAL)); await new Promise(r => setTimeout(r, READY_POLL_INTERVAL));
} }
throw new Error('kiwix-serve did not become ready in time'); throw new Error('kiwix-serve did not become ready in time');
} }
async #zimFiles() { async #zimFiles() {
return (await fs.promises.readdir(this.#dir).catch(() => [])).filter(f => f.endsWith('.zim')); return (await fs.promises.readdir(this.#dir).catch(() => []))
.filter(f => f.endsWith('.zim'));
}
#watchDir() {
this.#watcher?.close();
this.#watcher = fs.watch(this.#dir, (_event, filename) => {
if(!filename?.endsWith('.zim')) return;
clearTimeout(this.#watchTimer);
this.#watchTimer = setTimeout(() => this.#rebuildLibrary().catch(() => {}), WATCH_DEBOUNCE);
});
this.#watcher.on('error', () => {});
}
#unwatchDir() {
clearTimeout(this.#watchTimer);
this.#watchTimer = null;
this.#watcher?.close();
this.#watcher = null;
} }
/** Rebuilds library.xml and starts kiwix-serve. Resolves once the server is responding. */
async start() { async start() {
if(this.#remote || this.#child) return; if(this.#remote || this.#child) return;
await fs.promises.mkdir(this.#dir, {recursive: true}); await fs.promises.mkdir(this.#dir, {recursive: true});
await this.#rebuildLibrary(); await this.#rebuildLibrary();
this.#port ??= await findFreePort(); this.#port ??= await findFreePort();
this.#child = spawn(this.#bin('kiwix-serve'), ['--library', '-i', this.#host, '-p', String(this.#port), this.#libraryPath], {stdio: 'ignore'}); this.#child = spawn(this.#bin('kiwix-serve'), [
this.#child.on('exit', () => { this.#child = null; }); '--library',
'--monitorLibrary',
'-i',
this.#host,
'-p',
String(this.#port),
this.#libraryPath,
], {stdio: 'ignore'});
this.#child.on('exit', () => {
this.#child = null;
this.#unwatchDir();
});
try { try {
await this.#waitUntilReady(); await this.#waitUntilReady();
@@ -119,16 +173,22 @@ export class KiwixServer {
await this.stop(); await this.stop();
throw e; throw e;
} }
this.#watchDir();
} }
/** Gracefully stops kiwix-serve, if we own it. No-op if attached to a remote instance. */
async stop() { async stop() {
this.#unwatchDir();
if(this.#remote || !this.#child) return; if(this.#remote || !this.#child) return;
const child = this.#child; const child = this.#child;
await new Promise(resolve => { await new Promise(resolve => {
child.once('exit', resolve); child.once('exit', resolve);
child.kill('SIGTERM'); child.kill('SIGTERM');
}); });
this.#child = null; this.#child = null;
} }
@@ -137,22 +197,23 @@ export class KiwixServer {
await this.start(); await this.start();
} }
/** Rebuilds library.xml from disk and restarts kiwix-serve. No-op if attached to a remote instance -
* whoever owns that process is responsible for reloading it. */
async reload() { async reload() {
if(this.#remote || !this.#child) return; if(this.#remote || !this.#child) return;
await this.restart(); await this.#rebuildLibrary();
} }
/** Local catalog listing - same flat shape as the online catalog (catalog.js), plus a `file` field. */
async list() { async list() {
this.#assertRunning(); this.#assertRunning();
const xml = await this.#fetchLibraryXml(); const xml = await this.#fetchLibraryXml();
if(!xml) return []; if(!xml) return [];
const entries = fromXml(xml); const entries = fromXml(xml);
return (entries?.library?.book || []).map(e => {
return makeArray(entries?.library?.book || []).map(e => {
const tags = e.tags.split(';'); const tags = e.tags.split(';');
const name = e.path.replaceAll('.zim', ''); const name = e.path.replace(/\.zim$/i, '');
return { return {
id: e.id, id: e.id,
title: e.title, title: e.title,
@@ -167,6 +228,7 @@ export class KiwixServer {
publisher: e.publisher, publisher: e.publisher,
articleCount: +e.articleCount || 0, articleCount: +e.articleCount || 0,
sizeMb: +(Number(e.size) / 1024).toFixed(1) || 0, sizeMb: +(Number(e.size) / 1024).toFixed(1) || 0,
file: e.path,
href: name, href: name,
icon: `data:${e.faviconMimetype || 'image/png'};base64,${e.favicon}`, icon: `data:${e.faviconMimetype || 'image/png'};base64,${e.favicon}`,
viewer: `${this.baseUrl}/content/${name}`, viewer: `${this.baseUrl}/content/${name}`,
@@ -174,58 +236,126 @@ export class KiwixServer {
}); });
} }
/** Splits a href ("zim/path/to/page") or a full content/viewer URL into {zim, path}. */
#splitHref(href) { #splitHref(href) {
const clean = href.replace(`${this.baseUrl}/content/`, '').replace(/^\/+/, ''); const clean = href.replace(`${this.baseUrl}/content/`, '').replace(/^\/+/, '');
const [zim, ...rest] = clean.split('/'); const [zim, ...rest] = clean.split('/');
return {zim, path: rest.join('/')}; return {zim, path: rest.join('/')};
} }
/** Builds a kiwix-serve content URL from a href (as returned by list()/search()), or from an already-built content/viewer URL. */
link(href) { link(href) {
this.#assertRunning(); this.#assertRunning();
const {zim, path} = this.#splitHref(href); const {zim, path} = this.#splitHref(href);
return `${this.baseUrl}/content/${zim}${path ? '/' + path : ''}`; return `${this.baseUrl}/content/${zim}${path ? '/' + path : ''}`;
} }
/** Fetches a single asset's raw bytes straight from kiwix-serve. */
async raw(href) { async raw(href) {
const res = await fetch(this.link(href)); const res = await fetch(this.link(href));
if(!res.ok) return null; if(!res.ok) return null;
return {mimetype: res.headers.get('content-type'), data: Buffer.from(await res.arrayBuffer())};
return {
mimetype: res.headers.get('content-type'),
data: Buffer.from(await res.arrayBuffer()),
};
} }
/** Fulltext search across every local ZIM via kiwix-serve's own xapian index */ async #rawSearch(termList, scoped, bookMap, limit) {
async search(terms, limit = 20) { const perBook = await Promise.all(scoped.map(async book => {
this.#assertRunning(); const params = new URLSearchParams({
const termList = String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean); pattern: termList.join(' '),
if (!termList.length) return []; format: 'xml',
pageLength: String(limit),
'books.name': book.href,
});
const params = new URLSearchParams({pattern: termList.join(' '), format: 'xml', pageLength: String(limit)});
const res = await fetch(`${this.baseUrl}/search?${params}`); const res = await fetch(`${this.baseUrl}/search?${params}`);
if(!res.ok) return []; if(!res.ok) return [];
const found = fromXml(await res.text())?.rss?.channel?.item || []; const found = fromXml(await res.text())?.rss?.channel?.item || [];
return makeArray(found).map(hit => {
const resultBook = bookMap.get(hit.book?.title) || book;
const prefix = `/content/${resultBook.href}/`;
const page = hit.link.startsWith(prefix)
? hit.link.slice(prefix.length)
: hit.link.replace(/^\/+/, '');
return {
id: resultBook.id,
title: hit.title,
page,
name: resultBook.name,
publisher: resultBook.publisher,
href: `${resultBook.href}/${page}`,
icon: resultBook.icon,
viewer: this.baseUrl + hit.link,
summary: hit.description,
xapianScore: +hit.score || 0,
};
}).filter(Boolean);
}));
return rrfMerge(perBook);
}
/** Fulltext search across local ZIMs with ranking, diversification and spelling correction. */
async search(terms, {limit = 20, sources = null} = {}) {
this.#assertRunning();
const termList = tokenize(terms);
if(!termList.length) return {results: [], spellcheck: null};
const books = await this.list(); const books = await this.list();
const bookMap = new Map(books.map(b => [b.title, b])); const bookMap = new Map(books.map(b => [b.title, b]));
return found.map(hit => { const scoped = sources?.length
const book = bookMap.get(hit.book.title); ? books.filter(b => sources.includes(b.name) || sources.includes(b.href))
if (!book) return null; : books;
const prefix = `/content/${book.href}/`;
const page = hit.link.startsWith(prefix) ? hit.link.slice(prefix.length) : hit.link.replace(/^\/+/, ''); if(!scoped.length) return {results: [], spellcheck: null};
let activeTerms = termList;
let hits = await this.#rawSearch(activeTerms, scoped, bookMap, limit);
let spellcheck = null;
if(!hits.length && !this.#remote) {
const vocabulary = new Set();
for(const book of scoped) {
if(!book.file) continue;
try {
for(const word of await titleVocabulary(path.join(this.#dir, book.file)))
vocabulary.add(word);
} catch {
// Unreadable ZIM - skip it, don't fail the whole search.
}
}
const corrected = suggestCorrection(termList, vocabulary);
if(corrected) {
const retry = await this.#rawSearch(corrected, scoped, bookMap, limit);
if(retry.length) {
hits = retry;
activeTerms = corrected;
spellcheck = {
from: termList.join(' '),
to: corrected.join(' '),
};
}
}
}
const ranked = rerank(hits, activeTerms);
return { return {
id: book.id, results: diversify(ranked, limit),
title: hit.title, spellcheck,
page,
name: book.name,
publisher: book.publisher,
href: `${book.href}/${page}`,
icon: book.icon,
viewer: this.baseUrl + hit.link,
summary: hit.description,
score: +hit.score || 0,
}; };
}).filter(Boolean);
} }
} }
+7 -10
View File
@@ -6,34 +6,31 @@ export function levenshtein(a, b) {
for(let j = 0; j <= n; j++) dp[0][j] = j; for(let j = 0; j <= n; j++) dp[0][j] = j;
for(let i = 1; i <= m; i++) { for(let i = 1; i <= m; i++) {
for(let j = 1; j <= n; j++) { for(let j = 1; j <= n; j++) {
dp[i][j] = a[i - 1] === b[j - 1] dp[i][j] = a[i - 1] === b[j - 1] ? dp[i - 1][j - 1] : 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
? dp[i - 1][j - 1]
: 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
} }
} }
return dp[m][n]; return dp[m][n];
} }
function scoreAgainst(text, term) { function scoreAgainst(text, term) {
if (text.includes(term)) return 1 - (text.length - term.length) / text.length * 0.3; if (!text.length || !term.length) return 0;
if (!text.length || !term.length || text[0] !== term[0]) return 0; if (text === term) return 1;
const dist = levenshtein(text, term); if (text.includes(term)) return 0.8;
const maxAllowed = Math.max(1, Math.ceil(term.length * 0.34)); const maxAllowed = Math.max(1, Math.ceil(term.length * 0.34));
if (Math.abs(text.length - term.length) > maxAllowed) return 0;
const dist = levenshtein(text, term);
if (dist > maxAllowed) return 0; if (dist > maxAllowed) return 0;
return 1 - dist / Math.max(text.length, term.length); return 1 - dist / Math.max(text.length, term.length);
} }
/** Compares `target` against one or more search terms; returns avg/max/per-term similarity. */
export function fuzzyMatch(target, ...terms) { export function fuzzyMatch(target, ...terms) {
if (!terms.length) throw new Error('Requires at least 1 term to compare'); if (!terms.length) throw new Error('Requires at least 1 term to compare');
const lowerTarget = String(target).toLowerCase(); const lowerTarget = String(target).toLowerCase();
const words = lowerTarget.split(/\W+/).filter(Boolean); const words = lowerTarget.split(/\W+/).filter(Boolean);
const similarities = terms.map(term => { const similarities = terms.map(term => {
const t = term.toLowerCase(); const t = String(term).toLowerCase();
return Math.max(scoreAgainst(lowerTarget, t), ...words.map(w => scoreAgainst(w, t))); return Math.max(scoreAgainst(lowerTarget, t), ...words.map(w => scoreAgainst(w, t)));
}); });
return { return {
avg: similarities.reduce((acc, s) => acc + s, 0) / similarities.length, avg: similarities.reduce((acc, s) => acc + s, 0) / similarities.length,
max: Math.max(...similarities), max: Math.max(...similarities),