generated from ztimson/template
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8d4258b951 |
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@ztimson/zim-utils",
|
||||
"version": "0.3.8",
|
||||
"version": "0.4.0",
|
||||
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
|
||||
"author": "Zak Timson",
|
||||
"license": "MIT",
|
||||
|
||||
+3
-3
@@ -107,7 +107,7 @@ export class ZimManager {
|
||||
const file = match.href + (match.href.endsWith('.zim') ? '' : '.zim');
|
||||
await fs.promises.rm(path.join(this.#dir, file), {force: true});
|
||||
await this.#ensureServer();
|
||||
await this.#waitUntilVisible(file, {expect: false});
|
||||
await this.#waitUntilVicsible(file, {expect: false});
|
||||
return {file: match.file, status: 'deleted'};
|
||||
}
|
||||
|
||||
@@ -154,9 +154,9 @@ export class ZimManager {
|
||||
}
|
||||
|
||||
/** Two-pass fulltext search across every local ZIM, returns enriched results matching catalog/list shape. */
|
||||
async search(terms, limit = 20) {
|
||||
async search(terms, {limit = 20, sources = null} = {}) {
|
||||
const server = await this.#ensureServer();
|
||||
return server.search(terms, limit);
|
||||
return server.search(terms, {limit, sources});
|
||||
}
|
||||
|
||||
/** Checks all local ZIMs against the catalog and updates any that are outdated. */
|
||||
|
||||
+51
-4
@@ -6,6 +6,8 @@ import {decompressPool} from './decompress.js';
|
||||
const HEADER_SIZE = 80;
|
||||
const NS_CONTENT = 'C';
|
||||
const NS_METADATA = 'M';
|
||||
const VOCAB_CACHE_LIMIT = 4;
|
||||
const vocabCache = new Map(); // zimPath -> {mtimeMs, size, words: Set<string>};
|
||||
|
||||
async function readAt(fd, pos, length) {
|
||||
const buf = Buffer.alloc(length);
|
||||
@@ -33,7 +35,10 @@ async function readMimeTypes(fd, mimeListPos) {
|
||||
for (;;) {
|
||||
str += (await readAt(fd, pos, 1024)).toString('binary');
|
||||
const end = str.indexOf('\0\0');
|
||||
if (end !== -1) { str = str.slice(0, end + 1); break; }
|
||||
if (end !== -1) {
|
||||
str = str.slice(0, end + 1);
|
||||
break;
|
||||
}
|
||||
pos += 1024;
|
||||
}
|
||||
return str.split('\0').filter(Boolean);
|
||||
@@ -50,8 +55,15 @@ async function readDirent(fd, offset) {
|
||||
o += 4; // revision, unused
|
||||
|
||||
let redirectIndex = null, cluster = null, blob = null;
|
||||
if (mimetype === 0xffff) { redirectIndex = buf.readUInt32LE(o); o += 4; }
|
||||
else { cluster = buf.readUInt32LE(o); o += 4; blob = buf.readUInt32LE(o); o += 4; }
|
||||
if (mimetype === 0xffff) {
|
||||
redirectIndex = buf.readUInt32LE(o);
|
||||
o += 4;
|
||||
} else {
|
||||
cluster = buf.readUInt32LE(o);
|
||||
o += 4;
|
||||
blob = buf.readUInt32LE(o);
|
||||
o += 4;
|
||||
}
|
||||
|
||||
const urlEnd = buf.indexOf(0, o);
|
||||
if (urlEnd === -1) continue;
|
||||
@@ -73,7 +85,8 @@ async function findByUrl(fd, header, url, namespace) {
|
||||
const dirKey = dirent.namespace + dirent.url;
|
||||
const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0;
|
||||
if (cmp === 0) return dirent;
|
||||
if (cmp < 0) hi = mid - 1; else lo = mid + 1;
|
||||
if (cmp < 0) hi = mid - 1;
|
||||
else lo = mid + 1;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -136,3 +149,37 @@ export async function readZimMetadata(zimPath) {
|
||||
author: creator, publisher,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Unique, lowercased words pulled from every article title in a ZIM.
|
||||
* Only the deduped Set is cached (not the raw title list), and only for the
|
||||
* last VOCAB_CACHE_LIMIT ZIMs touched (evicted least-recently-used), so memory scales with recent search activity rather than total library size.
|
||||
* First call per ZIM does a full O(articleCount) directory walk; cached after that until the file's mtime/size changes.
|
||||
*/
|
||||
export async function titleVocabulary(zimPath) {
|
||||
const stat = await fs.promises.stat(zimPath);
|
||||
const cached = vocabCache.get(zimPath);
|
||||
if (cached && cached.mtimeMs === stat.mtimeMs && cached.size === stat.size) {
|
||||
vocabCache.delete(zimPath);
|
||||
vocabCache.set(zimPath, cached); // bump to most-recently-used
|
||||
return cached.words;
|
||||
}
|
||||
|
||||
const fd = await fs.promises.open(zimPath, 'r');
|
||||
let words;
|
||||
try {
|
||||
const header = await readHeader(fd);
|
||||
words = new Set();
|
||||
for (let i = 0; i < header.articleCount; i++) {
|
||||
const dirent = await readDirent(fd, await ptr64(fd, header.urlPtrPos, i));
|
||||
if (dirent.namespace !== NS_CONTENT || dirent.mimetype === 0xffff) continue; // skip redirects/non-content
|
||||
for (const w of (dirent.title || dirent.url).toLowerCase().split(/\W+/)) if (w) words.add(w);
|
||||
}
|
||||
} finally {
|
||||
await fd.close();
|
||||
}
|
||||
|
||||
vocabCache.set(zimPath, {mtimeMs: stat.mtimeMs, size: stat.size, words});
|
||||
if (vocabCache.size > VOCAB_CACHE_LIMIT) vocabCache.delete(vocabCache.keys().next().value);
|
||||
return words;
|
||||
}
|
||||
|
||||
+225
@@ -0,0 +1,225 @@
|
||||
'use strict';
|
||||
|
||||
import {fuzzyMatch} from './utils.js';
|
||||
|
||||
const RRF_K = 60;
|
||||
|
||||
/** Merges independently-ranked result lists without comparing their raw Xapian scores. */
|
||||
export function rrfMerge(groups) {
|
||||
const merged = new Map();
|
||||
|
||||
for(const group of groups) {
|
||||
group.forEach((hit, rank) => {
|
||||
const contribution = 1 / (RRF_K + rank + 1);
|
||||
const entry = merged.get(hit.href);
|
||||
|
||||
if(entry) entry.score += contribution;
|
||||
else merged.set(hit.href, {hit, score: contribution});
|
||||
});
|
||||
}
|
||||
|
||||
return [...merged.values()]
|
||||
.sort((a, b) => b.score - a.score)
|
||||
.map(({hit, score}) => ({...hit, rrf: score}));
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculates field relevance from exact terms, phrase matches, proximity,
|
||||
* match density and optionally fuzzy similarity.
|
||||
*/
|
||||
function fieldScore(text, terms, {fuzzy = false} = {}) {
|
||||
const value = String(text || '').trim().toLowerCase();
|
||||
|
||||
if(!value || !terms.length) {
|
||||
return {
|
||||
exact: 0,
|
||||
coverage: 0,
|
||||
proximity: 0,
|
||||
density: 0,
|
||||
fuzzy: 0,
|
||||
phrase: 0,
|
||||
score: 0,
|
||||
};
|
||||
}
|
||||
|
||||
const words = value.split(/\W+/).filter(Boolean);
|
||||
const normalizedTerms = terms.map(t => t.toLowerCase());
|
||||
const phrase = normalizedTerms.join(' ');
|
||||
|
||||
const exactTerms = normalizedTerms.filter(t => words.includes(t));
|
||||
const substringTerms = normalizedTerms.filter(t => value.includes(t));
|
||||
|
||||
const coverage = exactTerms.length / normalizedTerms.length;
|
||||
const substringCoverage = substringTerms.length / normalizedTerms.length;
|
||||
const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage;
|
||||
const phraseScore = value.includes(phrase) ? 1 : 0;
|
||||
|
||||
let proximity = 0;
|
||||
|
||||
if(normalizedTerms.length > 1) {
|
||||
const positions = [];
|
||||
|
||||
for(const term of normalizedTerms) {
|
||||
const index = words.indexOf(term);
|
||||
if(index !== -1) positions.push(index);
|
||||
}
|
||||
|
||||
if(positions.length > 1) {
|
||||
const span = Math.max(...positions) - Math.min(...positions);
|
||||
proximity = 1 / Math.max(1, span);
|
||||
}
|
||||
}
|
||||
|
||||
const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0);
|
||||
const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25));
|
||||
const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0;
|
||||
|
||||
const score =
|
||||
phraseScore * 1 +
|
||||
exact * 0.8 +
|
||||
proximity * 0.35 +
|
||||
substringCoverage * 0.25 +
|
||||
density * 0.15 +
|
||||
fuzzyScore * 0.75;
|
||||
|
||||
return {
|
||||
exact,
|
||||
coverage,
|
||||
proximity,
|
||||
density,
|
||||
fuzzy: fuzzyScore,
|
||||
phrase: phraseScore,
|
||||
score,
|
||||
};
|
||||
}
|
||||
|
||||
/** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */
|
||||
export function rerank(hits, termList) {
|
||||
if(!termList.length) return hits;
|
||||
|
||||
return hits.map(hit => {
|
||||
const title = fieldScore(hit.title, termList, {fuzzy: true});
|
||||
const summary = fieldScore(hit.summary, termList);
|
||||
const titleLower = (hit.title || '').toLowerCase();
|
||||
const summaryLower = (hit.summary || '').toLowerCase();
|
||||
const phrase = termList.join(' ').toLowerCase();
|
||||
|
||||
const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0;
|
||||
const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0;
|
||||
|
||||
const allTitleTerms = termList.every(term =>
|
||||
titleLower.split(/\W+/).includes(term.toLowerCase())
|
||||
) ? 1 : 0;
|
||||
|
||||
const allSummaryTerms = termList.every(term =>
|
||||
summaryLower.includes(term.toLowerCase())
|
||||
) ? 1 : 0;
|
||||
|
||||
const finalScore =
|
||||
hit.rrf +
|
||||
title.score * 1.25 +
|
||||
summary.score * 0.35 +
|
||||
titleExactPhrase * 1.5 +
|
||||
allTitleTerms * 0.75 +
|
||||
summaryExactPhrase * 0.2 +
|
||||
allSummaryTerms * 0.15;
|
||||
|
||||
return {
|
||||
...hit,
|
||||
finalScore,
|
||||
ranking: {
|
||||
rrf: hit.rrf,
|
||||
title: title.score,
|
||||
summary: summary.score,
|
||||
phrase: titleExactPhrase,
|
||||
coverage: title.coverage,
|
||||
proximity: title.proximity,
|
||||
fuzzy: title.fuzzy,
|
||||
},
|
||||
};
|
||||
}).sort((a, b) => b.finalScore - a.finalScore);
|
||||
}
|
||||
|
||||
/** Round-robins results between source ZIMs so one archive cannot dominate the page. */
|
||||
export function diversify(hits, limit) {
|
||||
const byBook = new Map();
|
||||
|
||||
for(const hit of hits)
|
||||
(byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit);
|
||||
|
||||
for(const list of byBook.values())
|
||||
list.sort((a, b) => b.finalScore - a.finalScore);
|
||||
|
||||
const queues = [...byBook.values()];
|
||||
const out = [];
|
||||
|
||||
for(let i = 0; out.length < limit && queues.some(q => q.length); i++) {
|
||||
const queue = queues[i % queues.length];
|
||||
if(queue.length) out.push(queue.shift());
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
function bucketByFirstChar(words) {
|
||||
const buckets = new Map();
|
||||
|
||||
for(const word of words) {
|
||||
const key = word[0];
|
||||
(buckets.get(key) ?? buckets.set(key, []).get(key)).push(word);
|
||||
}
|
||||
|
||||
return buckets;
|
||||
}
|
||||
|
||||
/** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */
|
||||
export function bestFuzzyMatch(term, buckets) {
|
||||
const lower = term.toLowerCase();
|
||||
const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34));
|
||||
const candidates = new Set();
|
||||
|
||||
// Check every character so a typo in the first character doesn't eliminate the correct word.
|
||||
for(const char of lower) {
|
||||
const bucket = buckets.get(char);
|
||||
if(bucket) for(const candidate of bucket) candidates.add(candidate);
|
||||
}
|
||||
|
||||
let best = null;
|
||||
let bestScore = 0;
|
||||
|
||||
for(const candidate of candidates) {
|
||||
if(Math.abs(candidate.length - lower.length) > maxDistance) continue;
|
||||
|
||||
const score = fuzzyMatch(candidate, lower).max;
|
||||
|
||||
if(score > bestScore) {
|
||||
bestScore = score;
|
||||
best = candidate;
|
||||
}
|
||||
}
|
||||
|
||||
return best;
|
||||
}
|
||||
|
||||
/** Suggests corrected terms from the supplied title vocabulary. */
|
||||
export function suggestCorrection(termList, vocabulary) {
|
||||
if(!vocabulary.size) return null;
|
||||
|
||||
const buckets = bucketByFirstChar(vocabulary);
|
||||
let changed = false;
|
||||
|
||||
const corrected = termList.map(term => {
|
||||
if(vocabulary.has(term.toLowerCase())) return term;
|
||||
|
||||
const fix = bestFuzzyMatch(term, buckets);
|
||||
|
||||
if(fix && fix !== term.toLowerCase()) {
|
||||
changed = true;
|
||||
return fix;
|
||||
}
|
||||
|
||||
return term;
|
||||
});
|
||||
|
||||
return changed ? corrected : null;
|
||||
}
|
||||
+162
-63
@@ -7,10 +7,12 @@ import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import {fileURLToPath} from 'node:url';
|
||||
import {fromXml, makeArray} from '@ztimson/utils';
|
||||
import {titleVocabulary} from './reader.js';
|
||||
import {diversify, rerank, rrfMerge, suggestCorrection} from './search.js';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin'); // npm package root/bin - where bin/install.js drops the kiwix-tools binaries
|
||||
const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin');
|
||||
const READY_TIMEOUT = 10_000;
|
||||
const READY_POLL_INTERVAL = 100;
|
||||
const WATCH_DEBOUNCE = 300;
|
||||
@@ -26,9 +28,10 @@ function findFreePort() {
|
||||
});
|
||||
}
|
||||
|
||||
/** Owns a kiwix-serve process's full lifecycle: library.xml, start/stop/reload, content + search access.
|
||||
* Watches its own directory for .zim files appearing/disappearing (from *any* writer - itself, a remote-attached
|
||||
* ZimManager, a script, whatever) and keeps library.xml + the running kiwix-serve process in sync automatically. */
|
||||
function tokenize(terms) {
|
||||
return String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
|
||||
}
|
||||
|
||||
export class KiwixServer {
|
||||
static #empty = '<?xml version="1.0" encoding="UTF-8" ?>\n<library version="20110515"></library>\n';
|
||||
|
||||
@@ -38,7 +41,7 @@ export class KiwixServer {
|
||||
#binDir;
|
||||
#libraryPath;
|
||||
#child = null;
|
||||
#remote; // baseUrl string if attached to an externally-managed kiwix-serve, else null
|
||||
#remote;
|
||||
#watcher = null;
|
||||
#watchTimer = null;
|
||||
|
||||
@@ -46,11 +49,6 @@ export class KiwixServer {
|
||||
get running() { return !!this.#remote || !!this.#child; }
|
||||
get baseUrl() { return this.#remote || (this.#child ? `http://${this.#host}:${this.#port}` : null); }
|
||||
|
||||
/** @param {{port?: number, host?: string, binDir?: string, url?: string}} [opts]
|
||||
* url: attach to an already-running kiwix-serve (e.g. one started elsewhere in your codebase) instead of
|
||||
* spawning/owning one - start/stop become no-ops, and library.xml is read over HTTP instead of disk.
|
||||
* Whichever instance *does* own the process is responsible for watching the shared directory; an attached
|
||||
* instance doesn't need its own watcher; it just reads whatever the owner already reloaded. */
|
||||
constructor(dir, {port, host = '127.0.0.1', binDir = DEFAULT_BIN_DIR, url} = {}) {
|
||||
this.#dir = dir;
|
||||
this.#host = host;
|
||||
@@ -58,25 +56,29 @@ export class KiwixServer {
|
||||
this.#binDir = binDir;
|
||||
this.#libraryPath = path.join(dir, 'library.xml');
|
||||
this.#remote = url ? url.replace(/\/$/, '') : null;
|
||||
if (!this.#remote) this.#ensureLocalStore();
|
||||
|
||||
if(!this.#remote) this.#ensureLocalStore();
|
||||
}
|
||||
|
||||
#ensureLocalStore() {
|
||||
fs.mkdirSync(this.#dir, {recursive: true});
|
||||
if (!fs.existsSync(this.#libraryPath)) fs.writeFileSync(this.#libraryPath, KiwixServer.#empty);
|
||||
|
||||
if(!fs.existsSync(this.#libraryPath))
|
||||
fs.writeFileSync(this.#libraryPath, KiwixServer.#empty);
|
||||
}
|
||||
|
||||
#assertRunning() {
|
||||
if (!this.running) throw new Error('KiwixServer is not running - call start() first');
|
||||
if(!this.running) throw new Error('KiwixServer is not running - call start() first');
|
||||
}
|
||||
|
||||
#bin(name) {
|
||||
return path.join(this.#binDir, process.platform === 'win32' ? `${name}.exe` : name);
|
||||
}
|
||||
|
||||
/** Reads library.xml from disk if we own the server, or over HTTP if attached to a remote one. */
|
||||
async #fetchLibraryXml() {
|
||||
if (!this.#remote) return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => '');
|
||||
if(!this.#remote)
|
||||
return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => '');
|
||||
|
||||
try {
|
||||
const res = await fetch(`${this.#remote}/library.xml`);
|
||||
return res.ok ? await res.text() : '';
|
||||
@@ -85,44 +87,55 @@ export class KiwixServer {
|
||||
}
|
||||
}
|
||||
|
||||
/** Rebuilds library.xml from scratch by scanning `dir` for .zim files - no-op if attached to a remote server.
|
||||
* Rewriting the file bumps its mtime, which kiwix-serve (started with --monitorLibrary) picks up on its own
|
||||
* and hot-reloads without needing a restart. */
|
||||
async #rebuildLibrary() {
|
||||
if (this.#remote) return;
|
||||
if(this.#remote) return;
|
||||
|
||||
await fs.promises.rm(this.#libraryPath, {force: true});
|
||||
|
||||
const files = await this.#zimFiles();
|
||||
if (!files.length) return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty);
|
||||
for (const f of files) await execFileAsync(this.#bin('kiwix-manage'), [this.#libraryPath, 'add', path.join(this.#dir, f)]);
|
||||
|
||||
if(!files.length)
|
||||
return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty);
|
||||
|
||||
for(const file of files)
|
||||
await execFileAsync(this.#bin('kiwix-manage'), [
|
||||
this.#libraryPath,
|
||||
'add',
|
||||
path.join(this.#dir, file),
|
||||
]);
|
||||
}
|
||||
|
||||
async #waitUntilReady() {
|
||||
const deadline = Date.now() + READY_TIMEOUT;
|
||||
while (Date.now() < deadline) {
|
||||
|
||||
while(Date.now() < deadline) {
|
||||
try {
|
||||
await fetch(`http://${this.#host}:${this.#port}/`);
|
||||
return;
|
||||
} catch {}
|
||||
|
||||
await new Promise(r => setTimeout(r, READY_POLL_INTERVAL));
|
||||
}
|
||||
|
||||
throw new Error('kiwix-serve did not become ready in time');
|
||||
}
|
||||
|
||||
async #zimFiles() {
|
||||
return (await fs.promises.readdir(this.#dir).catch(() => [])).filter(f => f.endsWith('.zim'));
|
||||
return (await fs.promises.readdir(this.#dir).catch(() => []))
|
||||
.filter(f => f.endsWith('.zim'));
|
||||
}
|
||||
|
||||
/** Watches `dir` for .zim files being added/removed by anyone (this process, a remote-attached ZimManager,
|
||||
* a manual copy) and debounces a library rebuild so kiwix-serve's --monitorLibrary picks it up. Only runs
|
||||
* while we own the process - an attached instance has nothing local to watch on behalf of. */
|
||||
#watchDir() {
|
||||
this.#watcher?.close();
|
||||
|
||||
this.#watcher = fs.watch(this.#dir, (_event, filename) => {
|
||||
if (!filename?.endsWith('.zim')) return;
|
||||
if(!filename?.endsWith('.zim')) return;
|
||||
|
||||
clearTimeout(this.#watchTimer);
|
||||
this.#watchTimer = setTimeout(() => this.#rebuildLibrary().catch(() => {}), WATCH_DEBOUNCE);
|
||||
});
|
||||
this.#watcher.on('error', () => {}); // e.g. dir removed out from under us - just stop watching, don't crash the process
|
||||
|
||||
this.#watcher.on('error', () => {});
|
||||
}
|
||||
|
||||
#unwatchDir() {
|
||||
@@ -132,23 +145,31 @@ export class KiwixServer {
|
||||
this.#watcher = null;
|
||||
}
|
||||
|
||||
/** Rebuilds library.xml and starts kiwix-serve. Resolves once the server is responding.
|
||||
* Starts with --monitorLibrary so the process reloads itself whenever library.xml's mtime changes -
|
||||
* no restart needed for updates after this. */
|
||||
async start() {
|
||||
if (this.#remote || this.#child) return;
|
||||
if(this.#remote || this.#child) return;
|
||||
|
||||
await fs.promises.mkdir(this.#dir, {recursive: true});
|
||||
await this.#rebuildLibrary();
|
||||
this.#port ??= await findFreePort();
|
||||
|
||||
this.#child = spawn(this.#bin('kiwix-serve'),
|
||||
['--library', '--monitorLibrary', '-i', this.#host, '-p', String(this.#port), this.#libraryPath],
|
||||
{stdio: 'ignore'});
|
||||
this.#child.on('exit', () => { this.#child = null; this.#unwatchDir(); });
|
||||
this.#child = spawn(this.#bin('kiwix-serve'), [
|
||||
'--library',
|
||||
'--monitorLibrary',
|
||||
'-i',
|
||||
this.#host,
|
||||
'-p',
|
||||
String(this.#port),
|
||||
this.#libraryPath,
|
||||
], {stdio: 'ignore'});
|
||||
|
||||
this.#child.on('exit', () => {
|
||||
this.#child = null;
|
||||
this.#unwatchDir();
|
||||
});
|
||||
|
||||
try {
|
||||
await this.#waitUntilReady();
|
||||
} catch (e) {
|
||||
} catch(e) {
|
||||
await this.stop();
|
||||
throw e;
|
||||
}
|
||||
@@ -156,15 +177,18 @@ export class KiwixServer {
|
||||
this.#watchDir();
|
||||
}
|
||||
|
||||
/** Gracefully stops kiwix-serve, if we own it. No-op if attached to a remote instance. */
|
||||
async stop() {
|
||||
this.#unwatchDir();
|
||||
if (this.#remote || !this.#child) return;
|
||||
|
||||
if(this.#remote || !this.#child) return;
|
||||
|
||||
const child = this.#child;
|
||||
|
||||
await new Promise(resolve => {
|
||||
child.once('exit', resolve);
|
||||
child.kill('SIGTERM');
|
||||
});
|
||||
|
||||
this.#child = null;
|
||||
}
|
||||
|
||||
@@ -173,24 +197,23 @@ export class KiwixServer {
|
||||
await this.start();
|
||||
}
|
||||
|
||||
/** Forces an immediate library rebuild rather than waiting for the directory watcher's debounce to fire.
|
||||
* Not required for correctness (the watcher does this automatically for any writer), just a manual way to
|
||||
* skip the ~300ms wait when you already know something changed. No-op if attached to a remote instance -
|
||||
* whoever owns that process already reloads itself. */
|
||||
async reload() {
|
||||
if (this.#remote || !this.#child) return;
|
||||
if(this.#remote || !this.#child) return;
|
||||
await this.#rebuildLibrary();
|
||||
}
|
||||
|
||||
/** Local catalog listing - same flat shape as the online catalog (catalog.js), plus a `file` field. */
|
||||
async list() {
|
||||
this.#assertRunning();
|
||||
|
||||
const xml = await this.#fetchLibraryXml();
|
||||
if (!xml) return [];
|
||||
if(!xml) return [];
|
||||
|
||||
const entries = fromXml(xml);
|
||||
|
||||
return makeArray(entries?.library?.book || []).map(e => {
|
||||
const tags = e.tags.split(';');
|
||||
const name = e.path.replace('.zim', '');
|
||||
|
||||
return {
|
||||
id: e.id,
|
||||
title: e.title,
|
||||
@@ -205,6 +228,7 @@ export class KiwixServer {
|
||||
publisher: e.publisher,
|
||||
articleCount: +e.articleCount || 0,
|
||||
sizeMb: +(Number(e.size) / 1024).toFixed(1) || 0,
|
||||
file: e.path,
|
||||
href: name,
|
||||
icon: `data:${e.faviconMimetype || 'image/png'};base64,${e.favicon}`,
|
||||
viewer: `${this.baseUrl}/content/${name}`,
|
||||
@@ -212,46 +236,63 @@ export class KiwixServer {
|
||||
});
|
||||
}
|
||||
|
||||
/** Splits a href ("zim/path/to/page") or a full content/viewer URL into {zim, path}. */
|
||||
#splitHref(href) {
|
||||
const clean = href.replace(`${this.baseUrl}/content/`, '').replace(/^\/+/, '');
|
||||
const [zim, ...rest] = clean.split('/');
|
||||
return {zim, path: rest.join('/')};
|
||||
}
|
||||
|
||||
/** Builds a kiwix-serve content URL from a href (as returned by list()/search()), or from an already-built content/viewer URL. */
|
||||
link(href) {
|
||||
this.#assertRunning();
|
||||
|
||||
const {zim, path} = this.#splitHref(href);
|
||||
|
||||
return `${this.baseUrl}/content/${zim}${path ? '/' + path : ''}`;
|
||||
}
|
||||
|
||||
/** Fetches a single asset's raw bytes straight from kiwix-serve. */
|
||||
async raw(href) {
|
||||
const res = await fetch(this.link(href));
|
||||
|
||||
if(!res.ok) return null;
|
||||
return {mimetype: res.headers.get('content-type'), data: Buffer.from(await res.arrayBuffer())};
|
||||
|
||||
return {
|
||||
mimetype: res.headers.get('content-type'),
|
||||
data: Buffer.from(await res.arrayBuffer()),
|
||||
};
|
||||
}
|
||||
|
||||
/** Fulltext search across every local ZIM via kiwix-serve's own xapian index */
|
||||
async search(terms, limit = 20) {
|
||||
this.#assertRunning();
|
||||
const termList = String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
|
||||
if (!termList.length) return [];
|
||||
async #rawSearch(termList, scoped, bookMap, limit) {
|
||||
const groups = new Map();
|
||||
|
||||
for(const book of scoped) {
|
||||
const lang = book.language || '';
|
||||
(groups.get(lang) ?? groups.set(lang, []).get(lang)).push(book);
|
||||
}
|
||||
|
||||
const perGroup = await Promise.all([...groups.values()].map(async group => {
|
||||
const params = new URLSearchParams({
|
||||
pattern: termList.join(' '),
|
||||
format: 'xml',
|
||||
pageLength: String(limit),
|
||||
});
|
||||
|
||||
for(const book of group)
|
||||
params.append('books.name', book.name);
|
||||
|
||||
const params = new URLSearchParams({pattern: termList.join(' '), format: 'xml', pageLength: String(limit)});
|
||||
const res = await fetch(`${this.baseUrl}/search?${params}`);
|
||||
if (!res.ok) return [];
|
||||
if(!res.ok) return [];
|
||||
|
||||
const found = fromXml(await res.text())?.rss?.channel?.item || [];
|
||||
const books = await this.list();
|
||||
const bookMap = new Map(books.map(b => [b.title, b]));
|
||||
|
||||
return found.map(hit => {
|
||||
const book = bookMap.get(hit.book.title);
|
||||
if (!book) return null;
|
||||
if(!book) return null;
|
||||
|
||||
const prefix = `/content/${book.href}/`;
|
||||
const page = hit.link.startsWith(prefix) ? hit.link.slice(prefix.length) : hit.link.replace(/^\/+/, '');
|
||||
const page = hit.link.startsWith(prefix)
|
||||
? hit.link.slice(prefix.length)
|
||||
: hit.link.replace(/^\/+/, '');
|
||||
|
||||
return {
|
||||
id: book.id,
|
||||
title: hit.title,
|
||||
@@ -262,8 +303,66 @@ export class KiwixServer {
|
||||
icon: book.icon,
|
||||
viewer: this.baseUrl + hit.link,
|
||||
summary: hit.description,
|
||||
score: +hit.score || 0,
|
||||
xapianScore: +hit.score || 0,
|
||||
};
|
||||
}).filter(Boolean).sort((a, b) => b.xapianScore - a.xapianScore);
|
||||
}));
|
||||
|
||||
return rrfMerge(perGroup);
|
||||
}
|
||||
|
||||
/** Fulltext search across local ZIMs with ranking, diversification and spelling correction. */
|
||||
async search(terms, {limit = 20, sources = null} = {}) {
|
||||
this.#assertRunning();
|
||||
|
||||
const termList = tokenize(terms);
|
||||
if(!termList.length) return {results: [], spellcheck: null};
|
||||
|
||||
const books = await this.list();
|
||||
const bookMap = new Map(books.map(b => [b.title, b]));
|
||||
const scoped = sources?.length
|
||||
? books.filter(b => sources.includes(b.name) || sources.includes(b.href))
|
||||
: books;
|
||||
|
||||
if(!scoped.length) return {results: [], spellcheck: null};
|
||||
|
||||
let activeTerms = termList;
|
||||
let hits = await this.#rawSearch(activeTerms, scoped, bookMap, limit);
|
||||
let spellcheck = null;
|
||||
|
||||
if(!hits.length && !this.#remote) {
|
||||
const vocabulary = new Set();
|
||||
|
||||
for(const book of scoped) {
|
||||
if(!book.file) continue;
|
||||
|
||||
try {
|
||||
for(const word of await titleVocabulary(path.join(this.#dir, book.file)))
|
||||
vocabulary.add(word);
|
||||
} catch {
|
||||
// Unreadable ZIM - skip it, don't fail the whole search.
|
||||
}
|
||||
}
|
||||
|
||||
const corrected = suggestCorrection(termList, vocabulary);
|
||||
|
||||
if(corrected) {
|
||||
const retry = await this.#rawSearch(corrected, scoped, bookMap, limit);
|
||||
|
||||
if(retry.length) {
|
||||
hits = retry;
|
||||
activeTerms = corrected;
|
||||
spellcheck = {
|
||||
from: termList.join(' '),
|
||||
to: corrected.join(' '),
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
results: diversify(rerank(hits, activeTerms), limit),
|
||||
spellcheck,
|
||||
};
|
||||
}).filter(Boolean);
|
||||
}
|
||||
}
|
||||
|
||||
+12
-15
@@ -1,39 +1,36 @@
|
||||
export function levenshtein(a, b) {
|
||||
const m = a.length, n = b.length;
|
||||
if (!m) return n;
|
||||
if (!n) return m;
|
||||
if(!m) return n;
|
||||
if(!n) return m;
|
||||
const dp = Array.from({length: m + 1}, (_, i) => [i, ...Array(n).fill(0)]);
|
||||
for (let j = 0; j <= n; j++) dp[0][j] = j;
|
||||
for (let i = 1; i <= m; i++) {
|
||||
for (let j = 1; j <= n; j++) {
|
||||
dp[i][j] = a[i - 1] === b[j - 1]
|
||||
? dp[i - 1][j - 1]
|
||||
: 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
|
||||
for(let j = 0; j <= n; j++) dp[0][j] = j;
|
||||
for(let i = 1; i <= m; i++) {
|
||||
for(let j = 1; j <= n; j++) {
|
||||
dp[i][j] = a[i - 1] === b[j - 1] ? dp[i - 1][j - 1] : 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
|
||||
}
|
||||
}
|
||||
return dp[m][n];
|
||||
}
|
||||
|
||||
function scoreAgainst(text, term) {
|
||||
if (text.includes(term)) return 1 - (text.length - term.length) / text.length * 0.3;
|
||||
if (!text.length || !term.length || text[0] !== term[0]) return 0;
|
||||
const dist = levenshtein(text, term);
|
||||
if (!text.length || !term.length) return 0;
|
||||
if (text === term) return 1;
|
||||
if (text.includes(term)) return 0.8;
|
||||
const maxAllowed = Math.max(1, Math.ceil(term.length * 0.34));
|
||||
if (Math.abs(text.length - term.length) > maxAllowed) return 0;
|
||||
const dist = levenshtein(text, term);
|
||||
if (dist > maxAllowed) return 0;
|
||||
return 1 - dist / Math.max(text.length, term.length);
|
||||
}
|
||||
|
||||
/** Compares `target` against one or more search terms; returns avg/max/per-term similarity. */
|
||||
export function fuzzyMatch(target, ...terms) {
|
||||
if (!terms.length) throw new Error('Requires at least 1 term to compare');
|
||||
const lowerTarget = String(target).toLowerCase();
|
||||
const words = lowerTarget.split(/\W+/).filter(Boolean);
|
||||
|
||||
const similarities = terms.map(term => {
|
||||
const t = term.toLowerCase();
|
||||
const t = String(term).toLowerCase();
|
||||
return Math.max(scoreAgainst(lowerTarget, t), ...words.map(w => scoreAgainst(w, t)));
|
||||
});
|
||||
|
||||
return {
|
||||
avg: similarities.reduce((acc, s) => acc + s, 0) / similarities.length,
|
||||
max: Math.max(...similarities),
|
||||
|
||||
Reference in New Issue
Block a user