generated from ztimson/template
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0a7e87f6b7 | ||
|
|
e12c3e3cc6 | ||
|
|
8d4258b951 | ||
|
|
15f15e06f1 | ||
|
|
997bd6064d | ||
|
|
4b86fe8051 | ||
|
|
6e4d2c8ae7 | ||
|
|
11f90c74cb | ||
|
|
08ecefbe05 | ||
|
|
66c5cfbe18 | ||
|
|
ebd2078bd7 | ||
|
|
754c66ff5c | ||
|
|
fee24e97cd | ||
|
|
50c4b2ac21 | ||
|
|
499d83c9be | ||
|
|
0f8637f913 |
+148
-43
@@ -2,80 +2,185 @@ import fs from 'node:fs';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
import https from 'node:https';
|
||||
import {execFileSync} from 'node:child_process';
|
||||
import {fileURLToPath} from 'node:url';
|
||||
import AdmZip from 'adm-zip';
|
||||
import * as tar from 'tar';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
|
||||
const VERSION = '3.8.1';
|
||||
const BASE_URL = 'https://download.kiwix.org/release/kiwix-tools';
|
||||
const PLATFORM_MAP = {linux: 'linux', darwin: 'macos', win32: 'win'};
|
||||
const MAX_REDIRECTS = 5;
|
||||
const BIN_DIR = path.join(__dirname, '..', 'bin');
|
||||
const PLATFORM_MAP = {
|
||||
linux: {name: 'linux', ext: 'tar.gz'},
|
||||
darwin: {name: 'macos', ext: 'tar.gz'},
|
||||
win32: {name: 'win', ext: 'zip'}
|
||||
};
|
||||
const ARCH_MAP = {
|
||||
linux: {
|
||||
x64: 'x86_64',
|
||||
arm64: 'aarch64',
|
||||
arm: 'armhf',
|
||||
ia32: 'i586'
|
||||
},
|
||||
darwin: {
|
||||
x64: 'x86_64',
|
||||
arm64: 'arm64'
|
||||
},
|
||||
win32: {
|
||||
x64: 'x86_64',
|
||||
ia32: 'i686'
|
||||
}
|
||||
};
|
||||
|
||||
const BIN_DIR = path.join(__dirname, '..', 'bin'); // project root/bin
|
||||
|
||||
/** Download a file, following redirects. */
|
||||
function download(url, dest) {
|
||||
function download(url, dest, redirectsLeft = MAX_REDIRECTS) {
|
||||
return new Promise((resolve, reject) => {
|
||||
const file = fs.createWriteStream(dest);
|
||||
https.get(url, res => {
|
||||
if (res.statusCode >= 300 && res.statusCode < 400 && res.headers.location) {
|
||||
const agent = new https.Agent({maxFreeSockets: 20, maxTotalSockets: 50});
|
||||
|
||||
const cleanupAndReject = err => {
|
||||
file.close();
|
||||
return resolve(download(res.headers.location, dest));
|
||||
fs.unlink(dest, () => reject(err));
|
||||
};
|
||||
|
||||
const req = https.get(url, {agent}, res => {
|
||||
if (res.statusCode >= 300 && res.statusCode < 400 && res.headers.location) {
|
||||
res.resume();
|
||||
file.close();
|
||||
if (redirectsLeft <= 0) {
|
||||
reject(new Error(`Too many redirects while downloading ${url}`));
|
||||
return;
|
||||
}
|
||||
|
||||
const nextUrl = new URL(res.headers.location, url).toString();
|
||||
resolve(download(nextUrl, dest, redirectsLeft - 1));
|
||||
return;
|
||||
}
|
||||
|
||||
if (!res.statusCode || res.statusCode >= 400) {
|
||||
return reject(new Error(`Download failed: ${res.statusCode} ${res.statusMessage}`));
|
||||
res.resume();
|
||||
cleanupAndReject(new Error(`HTTP ${res.statusCode} - ${res.statusMessage}`));
|
||||
return;
|
||||
}
|
||||
|
||||
res.pipe(file);
|
||||
file.on('finish', () => file.close(resolve));
|
||||
}).on('error', err => fs.unlink(dest, () => reject(err)));
|
||||
file.on('finish', () => {
|
||||
file.close(resolve);
|
||||
});
|
||||
file.on('error', cleanupAndReject);
|
||||
});
|
||||
|
||||
req.on('error', cleanupAndReject);
|
||||
});
|
||||
}
|
||||
|
||||
/** Recursively chmod all files under a directory. */
|
||||
function chmodRecursive(dir, mode) {
|
||||
for (const entry of fs.readdirSync(dir, {withFileTypes: true})) {
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) chmodRecursive(full, mode);
|
||||
else fs.chmodSync(full, mode);
|
||||
async function flattenExtractedDir(srcDir, destDir) {
|
||||
let entries = await fs.promises.readdir(srcDir, {withFileTypes: true});
|
||||
|
||||
let sourceDir = srcDir;
|
||||
if (entries.length === 1 && entries[0].isDirectory()) {
|
||||
sourceDir = path.join(srcDir, entries[0].name);
|
||||
entries = await fs.promises.readdir(sourceDir, {withFileTypes: true});
|
||||
}
|
||||
|
||||
for (const entry of entries) {
|
||||
const src = path.join(sourceDir, entry.name);
|
||||
const dest = path.join(destDir, entry.name);
|
||||
|
||||
await fs.promises.cp(src, dest, {
|
||||
recursive: true,
|
||||
force: true
|
||||
});
|
||||
|
||||
await fs.promises.rm(src, {recursive: true, force: true});
|
||||
}
|
||||
}
|
||||
|
||||
async function downloadAndExtract() {
|
||||
const platformName = PLATFORM_MAP[process.platform];
|
||||
if (!platformName) throw new Error(`Unsupported platform: ${process.platform}`);
|
||||
async function extractTarGz(archivePath, destDir) {
|
||||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'kiwix-extract-'));
|
||||
|
||||
const ext = platformName === 'win' ? 'zip' : 'tar.gz';
|
||||
const archiveName = `kiwix-tools_${platformName}-x86_64-${VERSION}.${ext}`;
|
||||
await tar.extract({
|
||||
file: archivePath,
|
||||
cwd: tmpDir
|
||||
});
|
||||
|
||||
await flattenExtractedDir(tmpDir, destDir);
|
||||
await fs.promises.rm(tmpDir, {recursive: true, force: true});
|
||||
}
|
||||
|
||||
async function extractZip(archivePath, destDir) {
|
||||
const zip = new AdmZip(archivePath);
|
||||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'kiwix-extract-'));
|
||||
|
||||
await new Promise((resolve, reject) => {
|
||||
zip.extractAllToAsync(tmpDir, true, false, err => {
|
||||
if (err) reject(err);
|
||||
else resolve();
|
||||
});
|
||||
});
|
||||
|
||||
await flattenExtractedDir(tmpDir, destDir);
|
||||
await fs.promises.rm(tmpDir, {recursive: true, force: true});
|
||||
}
|
||||
|
||||
function resolveArchiveName() {
|
||||
const platformInfo = PLATFORM_MAP[process.platform];
|
||||
if (!platformInfo) {
|
||||
throw new Error(`Unsupported platform: ${process.platform}`);
|
||||
}
|
||||
|
||||
const archName = ARCH_MAP[process.platform]?.[process.arch];
|
||||
if (!archName) {
|
||||
const supported = Object.keys(ARCH_MAP[process.platform] || {}).join(', ');
|
||||
throw new Error(
|
||||
`Unsupported architecture "${process.arch}" for platform "${process.platform}". ` +
|
||||
`Supported architectures: ${supported}`
|
||||
);
|
||||
}
|
||||
|
||||
return {
|
||||
archiveName: `kiwix-tools_${platformInfo.name}-${archName}-${VERSION}.${platformInfo.ext}`,
|
||||
ext: platformInfo.ext
|
||||
};
|
||||
}
|
||||
|
||||
async function install() {
|
||||
const {archiveName, ext} = resolveArchiveName();
|
||||
const url = `${BASE_URL}/${archiveName}`;
|
||||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'kiwix-tools-'));
|
||||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'kiwix-install-'));
|
||||
const archivePath = path.join(tmpDir, archiveName);
|
||||
|
||||
console.log('Downloading kiwix-tools from:', url);
|
||||
console.log(`Platform: ${process.platform}/${process.arch}`);
|
||||
console.log(`Downloading kiwix-tools: ${url}`);
|
||||
await download(url, archivePath);
|
||||
|
||||
await fs.promises.mkdir(BIN_DIR, {recursive: true});
|
||||
console.log('Download complete!');
|
||||
|
||||
console.log('Extracting...');
|
||||
execFileSync('tar', ['-xf', archivePath, '-C', BIN_DIR]); // bsdtar handles zip too
|
||||
await fs.promises.mkdir(BIN_DIR, {recursive: true});
|
||||
|
||||
// Archives (tar.gz) may wrap contents in a subdirectory - flatten it into BIN_DIR
|
||||
const wrapperDir = (await fs.promises.readdir(BIN_DIR, {withFileTypes: true}))
|
||||
.find(e => e.isDirectory() && e.name.startsWith('kiwix-tools_'));
|
||||
if (wrapperDir) {
|
||||
const wrapperPath = path.join(BIN_DIR, wrapperDir.name);
|
||||
for (const entry of await fs.promises.readdir(wrapperPath)) {
|
||||
await fs.promises.rename(path.join(wrapperPath, entry), path.join(BIN_DIR, entry));
|
||||
}
|
||||
await fs.promises.rmdir(wrapperPath);
|
||||
if (ext === 'zip') {
|
||||
await extractZip(archivePath, BIN_DIR);
|
||||
} else {
|
||||
await extractTarGz(archivePath, BIN_DIR);
|
||||
}
|
||||
|
||||
if (process.platform !== 'win32') chmodRecursive(BIN_DIR, 0o755);
|
||||
for (const entry of await fs.promises.readdir(BIN_DIR)) {
|
||||
if (entry.startsWith('kiwix-')) {
|
||||
const fullPath = path.join(BIN_DIR, entry);
|
||||
const stat = await fs.promises.stat(fullPath);
|
||||
|
||||
if (stat.isFile()) {
|
||||
await fs.promises.chmod(fullPath, 0o755);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`Installed to: ${BIN_DIR}`);
|
||||
await fs.promises.rm(tmpDir, {recursive: true, force: true});
|
||||
console.log('Installed to:', BIN_DIR);
|
||||
}
|
||||
|
||||
downloadAndExtract().catch(err => {
|
||||
console.error(err);
|
||||
process.exit(1)
|
||||
}).finally(() => process.exit());
|
||||
install().catch(err => {
|
||||
console.error('Install failed:', err.message);
|
||||
process.exit(1);
|
||||
});
|
||||
|
||||
Generated
+80
-2
@@ -1,20 +1,34 @@
|
||||
{
|
||||
"name": "@ztimson/zim-utils",
|
||||
"version": "0.2.5",
|
||||
"version": "0.3.6",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@ztimson/zim-utils",
|
||||
"version": "0.2.5",
|
||||
"version": "0.3.6",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@ztimson/utils": "^0.30.8",
|
||||
"adm-zip": "^0.5.16",
|
||||
"lzma1": "^0.3.0",
|
||||
"tar": "^7.4.3",
|
||||
"zstd-codec": "^0.1.5"
|
||||
}
|
||||
},
|
||||
"node_modules/@isaacs/fs-minipass": {
|
||||
"version": "4.0.1",
|
||||
"resolved": "https://registry.npmjs.org/@isaacs/fs-minipass/-/fs-minipass-4.0.1.tgz",
|
||||
"integrity": "sha512-wgm9Ehl2jpeqP3zw/7mo3kRHFp5MEDhqAdwy1fTGkHAwnkGOVsgpvQhL8B5n1qlb01jV3n/bI0ZfZp5lWA1k4w==",
|
||||
"license": "ISC",
|
||||
"dependencies": {
|
||||
"minipass": "^7.0.4"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@ztimson/utils": {
|
||||
"version": "0.30.8",
|
||||
"resolved": "https://registry.npmjs.org/@ztimson/utils/-/utils-0.30.8.tgz",
|
||||
@@ -24,6 +38,24 @@
|
||||
"var-persist": "^1.0.1"
|
||||
}
|
||||
},
|
||||
"node_modules/adm-zip": {
|
||||
"version": "0.5.18",
|
||||
"resolved": "https://registry.npmjs.org/adm-zip/-/adm-zip-0.5.18.tgz",
|
||||
"integrity": "sha512-ufJnssQGbxzLNS1Ho9bCtX4rQKCCvoVuDLHoJyc3F9dOGDB4BkWs2Ci0kv53lqocAEQ/Cbi+I2XCsNYGqVYqng==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=12.0"
|
||||
}
|
||||
},
|
||||
"node_modules/chownr": {
|
||||
"version": "3.0.0",
|
||||
"resolved": "https://registry.npmjs.org/chownr/-/chownr-3.0.0.tgz",
|
||||
"integrity": "sha512-+IxzY9BZOQd/XuYPRmrvEVjF/nqj5kgT4kEq7VofrDoM1MxoRjEWkrCC3EtLi59TVawxTAn+orJwFQcrqEN1+g==",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/lzma1": {
|
||||
"version": "0.3.0",
|
||||
"resolved": "https://registry.npmjs.org/lzma1/-/lzma1-0.3.0.tgz",
|
||||
@@ -37,12 +69,58 @@
|
||||
"url": "https://github.com/sponsors/xseman"
|
||||
}
|
||||
},
|
||||
"node_modules/minipass": {
|
||||
"version": "7.1.3",
|
||||
"resolved": "https://registry.npmjs.org/minipass/-/minipass-7.1.3.tgz",
|
||||
"integrity": "sha512-tEBHqDnIoM/1rXME1zgka9g6Q2lcoCkxHLuc7ODJ5BxbP5d4c2Z5cGgtXAku59200Cx7diuHTOYfSBD8n6mm8A==",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"engines": {
|
||||
"node": ">=16 || 14 >=14.17"
|
||||
}
|
||||
},
|
||||
"node_modules/minizlib": {
|
||||
"version": "3.1.0",
|
||||
"resolved": "https://registry.npmjs.org/minizlib/-/minizlib-3.1.0.tgz",
|
||||
"integrity": "sha512-KZxYo1BUkWD2TVFLr0MQoM8vUUigWD3LlD83a/75BqC+4qE0Hb1Vo5v1FgcfaNXvfXzr+5EhQ6ing/CaBijTlw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"minipass": "^7.1.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">= 18"
|
||||
}
|
||||
},
|
||||
"node_modules/tar": {
|
||||
"version": "7.5.22",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.22.tgz",
|
||||
"integrity": "sha512-MFO/QzvtAOmJbkhOaCTvbGcFN9L9b+JunIsDwaKljSOdcLMea3NJ1k9Usz/rjdfSXTq4dfzfeS7W4p4YOAAHeA==",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"dependencies": {
|
||||
"@isaacs/fs-minipass": "^4.0.0",
|
||||
"chownr": "^3.0.0",
|
||||
"minipass": "^7.1.2",
|
||||
"minizlib": "^3.1.0",
|
||||
"yallist": "^5.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/var-persist": {
|
||||
"version": "1.0.1",
|
||||
"resolved": "https://registry.npmjs.org/var-persist/-/var-persist-1.0.1.tgz",
|
||||
"integrity": "sha512-Zon+pwvEpb0dEQCVShoMQQWV1JbWi4P2knW3h2sfSZS3pLecgbFig76tMSHEECQuEQ3KfYhMXgRDDIybtHTyZw==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/yallist": {
|
||||
"version": "5.0.0",
|
||||
"resolved": "https://registry.npmjs.org/yallist/-/yallist-5.0.0.tgz",
|
||||
"integrity": "sha512-YgvUTfwqyc7UXVMrB+SImsVYSmTS8X/tSrtdNZMImM+n7+QTriRXyXim0mBrTXNeqzVF0KWGgHPeiyViFFrNDw==",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/zstd-codec": {
|
||||
"version": "0.1.5",
|
||||
"resolved": "https://registry.npmjs.org/zstd-codec/-/zstd-codec-0.1.5.tgz",
|
||||
|
||||
+4
-2
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@ztimson/zim-utils",
|
||||
"version": "0.3.4",
|
||||
"version": "0.4.2",
|
||||
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
|
||||
"author": "Zak Timson",
|
||||
"license": "MIT",
|
||||
@@ -17,6 +17,8 @@
|
||||
"dependencies": {
|
||||
"@ztimson/utils": "^0.30.8",
|
||||
"lzma1": "^0.3.0",
|
||||
"zstd-codec": "^0.1.5"
|
||||
"tar": "^7.4.3",
|
||||
"zstd-codec": "^0.1.5",
|
||||
"adm-zip": "^0.5.16"
|
||||
}
|
||||
}
|
||||
|
||||
+35
-11
@@ -5,8 +5,13 @@ import {Readable} from 'node:stream';
|
||||
import {KiwixServer} from './server.js';
|
||||
import {zimCatalog, zimCatalogInfo, CATALOG_URL} from './catalog.js';
|
||||
|
||||
const VISIBLE_TIMEOUT = 5_000;
|
||||
const VISIBLE_POLL_INTERVAL = 200;
|
||||
|
||||
/** Manages a local directory of ZIM archives: catalog search, downloads, update checks, deletion.
|
||||
* Reuses (or owns & lazily starts) a KiwixServer for local listing/search, so `list()` and `catalog()` return the same shape. */
|
||||
* Reuses (or owns & lazily starts) a KiwixServer for local listing/search, so `list()` and `catalog()` return the same shape.
|
||||
* Doesn't drive library.xml/kiwix-serve reloads itself - the owning KiwixServer watches its directory and
|
||||
* reloads automatically whenever .zim files change, no matter which ZimManager (local or remote-attached) wrote them. */
|
||||
export class ZimManager {
|
||||
#catalogUrl;
|
||||
#dir;
|
||||
@@ -24,6 +29,10 @@ export class ZimManager {
|
||||
this.#server = server ?? new KiwixServer(dir, {port, host, binDir, url});
|
||||
}
|
||||
|
||||
#stripDate(filename) {
|
||||
return filename.replace(/\.zim$/i, '').replace(/_\d{4}-\d{2}(?:_\d+)?$/, '');
|
||||
}
|
||||
|
||||
/** The KiwixServer backing this manager - reuse it directly for content/search access, or pass into another ZimManager. */
|
||||
get server() { return this.#server; }
|
||||
|
||||
@@ -52,6 +61,20 @@ export class ZimManager {
|
||||
return {res, url: m[1]};
|
||||
}
|
||||
|
||||
/** Polls list() until `filename` shows up (or disappears, if `expect: false`), so callers get an accurate
|
||||
* status back rather than one that's ahead of what the server has actually picked up yet. The owning
|
||||
* KiwixServer's directory watcher does the real reload work in the background; this just waits for it. */
|
||||
async #waitUntilVisible(filename, {expect = true, timeout = VISIBLE_TIMEOUT} = {}) {
|
||||
const deadline = Date.now() + timeout;
|
||||
while (Date.now() < deadline) {
|
||||
const local = await this.list();
|
||||
const present = local.some(l => l.file === filename);
|
||||
if (present === expect) return true;
|
||||
await new Promise(r => setTimeout(r, VISIBLE_POLL_INTERVAL));
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
async #update(name, catalogEntry, localMatch, force) {
|
||||
const remoteDate = catalogEntry.updated ? new Date(catalogEntry.updated) : null;
|
||||
const localDate = localMatch?.updated ?? null;
|
||||
@@ -63,8 +86,8 @@ export class ZimManager {
|
||||
const destPath = path.join(this.#dir, filename);
|
||||
await this.#download(catalogEntry.href, destPath);
|
||||
if (localMatch && localMatch.file !== filename) await fs.promises.rm(path.join(this.#dir, localMatch.file), {force: true});
|
||||
const server = await this.#ensureServer();
|
||||
await server.reload();
|
||||
await this.#ensureServer();
|
||||
await this.#waitUntilVisible(filename, {expect: true});
|
||||
return {name, status: 'updated', file: filename};
|
||||
}
|
||||
|
||||
@@ -83,21 +106,22 @@ export class ZimManager {
|
||||
if (!match) throw new Error(`ZIM not found locally: ${nameOrFile}`);
|
||||
const file = match.href + (match.href.endsWith('.zim') ? '' : '.zim');
|
||||
await fs.promises.rm(path.join(this.#dir, file), {force: true});
|
||||
const server = await this.#ensureServer();
|
||||
await server.reload();
|
||||
await this.#ensureServer();
|
||||
await this.#waitUntilVisible(file, {expect: false});
|
||||
return {file: match.file, status: 'deleted'};
|
||||
}
|
||||
|
||||
/** Downloads/updates a single ZIM by direct href, matching against any existing local copy by name. */
|
||||
async download(href, {force = false} = {}) {
|
||||
await fs.promises.mkdir(this.#dir, {recursive: true});
|
||||
const {url: finalUrl} = await this.#resolveUrl(href);
|
||||
const filename = path.basename(new URL(finalUrl).pathname).replace(/\.meta4$/i, '');
|
||||
const name = filename.replace(/\.zim$/i, '').replace(/_\d{4}-\d{2}(?:_\d+)?$/, '');
|
||||
|
||||
await fs.promises.mkdir(this.#dir, {recursive: true});
|
||||
const local = await this.list();
|
||||
const exact = local.find(l => l.file === filename);
|
||||
if (exact && !force) return {name: exact.name ?? this.#stripDate(filename), status: 'skipped', reason: 'already downloaded', file: filename};
|
||||
const name = this.#stripDate(filename);
|
||||
const localMatch = local.find(l => l.name === name) ?? null;
|
||||
const catalogEntry = await zimCatalogInfo(name, this.#catalogUrl) || {name, updated: null, href: href};
|
||||
const catalogEntry = await zimCatalogInfo(name, this.#catalogUrl) || {name, updated: null, href};
|
||||
return this.#update(name, catalogEntry, localMatch, force);
|
||||
}
|
||||
|
||||
@@ -130,9 +154,9 @@ export class ZimManager {
|
||||
}
|
||||
|
||||
/** Two-pass fulltext search across every local ZIM, returns enriched results matching catalog/list shape. */
|
||||
async search(terms, limit = 20) {
|
||||
async search(terms, {limit = 20, sources = null} = {}) {
|
||||
const server = await this.#ensureServer();
|
||||
return server.search(terms, limit);
|
||||
return server.search(terms, {limit, sources});
|
||||
}
|
||||
|
||||
/** Checks all local ZIMs against the catalog and updates any that are outdated. */
|
||||
|
||||
+51
-4
@@ -6,6 +6,8 @@ import {decompressPool} from './decompress.js';
|
||||
const HEADER_SIZE = 80;
|
||||
const NS_CONTENT = 'C';
|
||||
const NS_METADATA = 'M';
|
||||
const VOCAB_CACHE_LIMIT = 4;
|
||||
const vocabCache = new Map(); // zimPath -> {mtimeMs, size, words: Set<string>};
|
||||
|
||||
async function readAt(fd, pos, length) {
|
||||
const buf = Buffer.alloc(length);
|
||||
@@ -33,7 +35,10 @@ async function readMimeTypes(fd, mimeListPos) {
|
||||
for (;;) {
|
||||
str += (await readAt(fd, pos, 1024)).toString('binary');
|
||||
const end = str.indexOf('\0\0');
|
||||
if (end !== -1) { str = str.slice(0, end + 1); break; }
|
||||
if (end !== -1) {
|
||||
str = str.slice(0, end + 1);
|
||||
break;
|
||||
}
|
||||
pos += 1024;
|
||||
}
|
||||
return str.split('\0').filter(Boolean);
|
||||
@@ -50,8 +55,15 @@ async function readDirent(fd, offset) {
|
||||
o += 4; // revision, unused
|
||||
|
||||
let redirectIndex = null, cluster = null, blob = null;
|
||||
if (mimetype === 0xffff) { redirectIndex = buf.readUInt32LE(o); o += 4; }
|
||||
else { cluster = buf.readUInt32LE(o); o += 4; blob = buf.readUInt32LE(o); o += 4; }
|
||||
if (mimetype === 0xffff) {
|
||||
redirectIndex = buf.readUInt32LE(o);
|
||||
o += 4;
|
||||
} else {
|
||||
cluster = buf.readUInt32LE(o);
|
||||
o += 4;
|
||||
blob = buf.readUInt32LE(o);
|
||||
o += 4;
|
||||
}
|
||||
|
||||
const urlEnd = buf.indexOf(0, o);
|
||||
if (urlEnd === -1) continue;
|
||||
@@ -73,7 +85,8 @@ async function findByUrl(fd, header, url, namespace) {
|
||||
const dirKey = dirent.namespace + dirent.url;
|
||||
const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0;
|
||||
if (cmp === 0) return dirent;
|
||||
if (cmp < 0) hi = mid - 1; else lo = mid + 1;
|
||||
if (cmp < 0) hi = mid - 1;
|
||||
else lo = mid + 1;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -136,3 +149,37 @@ export async function readZimMetadata(zimPath) {
|
||||
author: creator, publisher,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Unique, lowercased words pulled from every article title in a ZIM.
|
||||
* Only the deduped Set is cached (not the raw title list), and only for the
|
||||
* last VOCAB_CACHE_LIMIT ZIMs touched (evicted least-recently-used), so memory scales with recent search activity rather than total library size.
|
||||
* First call per ZIM does a full O(articleCount) directory walk; cached after that until the file's mtime/size changes.
|
||||
*/
|
||||
export async function titleVocabulary(zimPath) {
|
||||
const stat = await fs.promises.stat(zimPath);
|
||||
const cached = vocabCache.get(zimPath);
|
||||
if (cached && cached.mtimeMs === stat.mtimeMs && cached.size === stat.size) {
|
||||
vocabCache.delete(zimPath);
|
||||
vocabCache.set(zimPath, cached); // bump to most-recently-used
|
||||
return cached.words;
|
||||
}
|
||||
|
||||
const fd = await fs.promises.open(zimPath, 'r');
|
||||
let words;
|
||||
try {
|
||||
const header = await readHeader(fd);
|
||||
words = new Set();
|
||||
for (let i = 0; i < header.articleCount; i++) {
|
||||
const dirent = await readDirent(fd, await ptr64(fd, header.urlPtrPos, i));
|
||||
if (dirent.namespace !== NS_CONTENT || dirent.mimetype === 0xffff) continue; // skip redirects/non-content
|
||||
for (const w of (dirent.title || dirent.url).toLowerCase().split(/\W+/)) if (w) words.add(w);
|
||||
}
|
||||
} finally {
|
||||
await fd.close();
|
||||
}
|
||||
|
||||
vocabCache.set(zimPath, {mtimeMs: stat.mtimeMs, size: stat.size, words});
|
||||
if (vocabCache.size > VOCAB_CACHE_LIMIT) vocabCache.delete(vocabCache.keys().next().value);
|
||||
return words;
|
||||
}
|
||||
|
||||
+225
@@ -0,0 +1,225 @@
|
||||
'use strict';
|
||||
|
||||
import {fuzzyMatch} from './utils.js';
|
||||
|
||||
const RRF_K = 60;
|
||||
|
||||
/** Merges independently-ranked result lists without comparing their raw Xapian scores. */
|
||||
export function rrfMerge(groups) {
|
||||
const merged = new Map();
|
||||
|
||||
for(const group of groups) {
|
||||
group.forEach((hit, rank) => {
|
||||
const contribution = 1 / (RRF_K + rank + 1);
|
||||
const entry = merged.get(hit.href);
|
||||
|
||||
if(entry) entry.score += contribution;
|
||||
else merged.set(hit.href, {hit, score: contribution});
|
||||
});
|
||||
}
|
||||
|
||||
return [...merged.values()]
|
||||
.sort((a, b) => b.score - a.score)
|
||||
.map(({hit, score}) => ({...hit, rrf: score}));
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculates field relevance from exact terms, phrase matches, proximity,
|
||||
* match density and optionally fuzzy similarity.
|
||||
*/
|
||||
function fieldScore(text, terms, {fuzzy = false} = {}) {
|
||||
const value = String(text || '').trim().toLowerCase();
|
||||
|
||||
if(!value || !terms.length) {
|
||||
return {
|
||||
exact: 0,
|
||||
coverage: 0,
|
||||
proximity: 0,
|
||||
density: 0,
|
||||
fuzzy: 0,
|
||||
phrase: 0,
|
||||
score: 0,
|
||||
};
|
||||
}
|
||||
|
||||
const words = value.split(/\W+/).filter(Boolean);
|
||||
const normalizedTerms = terms.map(t => t.toLowerCase());
|
||||
const phrase = normalizedTerms.join(' ');
|
||||
|
||||
const exactTerms = normalizedTerms.filter(t => words.includes(t));
|
||||
const substringTerms = normalizedTerms.filter(t => value.includes(t));
|
||||
|
||||
const coverage = exactTerms.length / normalizedTerms.length;
|
||||
const substringCoverage = substringTerms.length / normalizedTerms.length;
|
||||
const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage;
|
||||
const phraseScore = value.includes(phrase) ? 1 : 0;
|
||||
|
||||
let proximity = 0;
|
||||
|
||||
if(normalizedTerms.length > 1) {
|
||||
const positions = [];
|
||||
|
||||
for(const term of normalizedTerms) {
|
||||
const index = words.indexOf(term);
|
||||
if(index !== -1) positions.push(index);
|
||||
}
|
||||
|
||||
if(positions.length > 1) {
|
||||
const span = Math.max(...positions) - Math.min(...positions);
|
||||
proximity = 1 / Math.max(1, span);
|
||||
}
|
||||
}
|
||||
|
||||
const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0);
|
||||
const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25));
|
||||
const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0;
|
||||
|
||||
const score =
|
||||
phraseScore * 1 +
|
||||
exact * 0.8 +
|
||||
proximity * 0.35 +
|
||||
substringCoverage * 0.25 +
|
||||
density * 0.15 +
|
||||
fuzzyScore * 0.75;
|
||||
|
||||
return {
|
||||
exact,
|
||||
coverage,
|
||||
proximity,
|
||||
density,
|
||||
fuzzy: fuzzyScore,
|
||||
phrase: phraseScore,
|
||||
score,
|
||||
};
|
||||
}
|
||||
|
||||
/** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */
|
||||
export function rerank(hits, termList) {
|
||||
if(!termList.length) return hits;
|
||||
|
||||
return hits.map(hit => {
|
||||
const title = fieldScore(hit.title, termList, {fuzzy: true});
|
||||
const summary = fieldScore(hit.summary, termList);
|
||||
const titleLower = (hit.title || '').toLowerCase();
|
||||
const summaryLower = (hit.summary?._text || '').toLowerCase();
|
||||
const phrase = termList.join(' ').toLowerCase();
|
||||
|
||||
const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0;
|
||||
const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0;
|
||||
|
||||
const allTitleTerms = termList.every(term =>
|
||||
titleLower.split(/\W+/).includes(term.toLowerCase())
|
||||
) ? 1 : 0;
|
||||
|
||||
const allSummaryTerms = termList.every(term =>
|
||||
summaryLower.includes(term.toLowerCase())
|
||||
) ? 1 : 0;
|
||||
|
||||
const finalScore =
|
||||
hit.rrf +
|
||||
title.score * 1.25 +
|
||||
summary.score * 0.35 +
|
||||
titleExactPhrase * 1.5 +
|
||||
allTitleTerms * 0.75 +
|
||||
summaryExactPhrase * 0.2 +
|
||||
allSummaryTerms * 0.15;
|
||||
|
||||
return {
|
||||
...hit,
|
||||
finalScore,
|
||||
ranking: {
|
||||
rrf: hit.rrf,
|
||||
title: title.score,
|
||||
summary: summary.score,
|
||||
phrase: titleExactPhrase,
|
||||
coverage: title.coverage,
|
||||
proximity: title.proximity,
|
||||
fuzzy: title.fuzzy,
|
||||
},
|
||||
};
|
||||
}).sort((a, b) => b.finalScore - a.finalScore);
|
||||
}
|
||||
|
||||
/** Round-robins results between source ZIMs so one archive cannot dominate the page. */
|
||||
export function diversify(hits, limit) {
|
||||
const byBook = new Map();
|
||||
|
||||
for(const hit of hits)
|
||||
(byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit);
|
||||
|
||||
for(const list of byBook.values())
|
||||
list.sort((a, b) => b.finalScore - a.finalScore);
|
||||
|
||||
const queues = [...byBook.values()];
|
||||
const out = [];
|
||||
|
||||
for(let i = 0; out.length < limit && queues.some(q => q.length); i++) {
|
||||
const queue = queues[i % queues.length];
|
||||
if(queue.length) out.push(queue.shift());
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
function bucketByFirstChar(words) {
|
||||
const buckets = new Map();
|
||||
|
||||
for(const word of words) {
|
||||
const key = word[0];
|
||||
(buckets.get(key) ?? buckets.set(key, []).get(key)).push(word);
|
||||
}
|
||||
|
||||
return buckets;
|
||||
}
|
||||
|
||||
/** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */
|
||||
export function bestFuzzyMatch(term, buckets) {
|
||||
const lower = term.toLowerCase();
|
||||
const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34));
|
||||
const candidates = new Set();
|
||||
|
||||
// Check every character so a typo in the first character doesn't eliminate the correct word.
|
||||
for(const char of lower) {
|
||||
const bucket = buckets.get(char);
|
||||
if(bucket) for(const candidate of bucket) candidates.add(candidate);
|
||||
}
|
||||
|
||||
let best = null;
|
||||
let bestScore = 0;
|
||||
|
||||
for(const candidate of candidates) {
|
||||
if(Math.abs(candidate.length - lower.length) > maxDistance) continue;
|
||||
|
||||
const score = fuzzyMatch(candidate, lower).max;
|
||||
|
||||
if(score > bestScore) {
|
||||
bestScore = score;
|
||||
best = candidate;
|
||||
}
|
||||
}
|
||||
|
||||
return best;
|
||||
}
|
||||
|
||||
/** Suggests corrected terms from the supplied title vocabulary. */
|
||||
export function suggestCorrection(termList, vocabulary) {
|
||||
if(!vocabulary.size) return null;
|
||||
|
||||
const buckets = bucketByFirstChar(vocabulary);
|
||||
let changed = false;
|
||||
|
||||
const corrected = termList.map(term => {
|
||||
if(vocabulary.has(term.toLowerCase())) return term;
|
||||
|
||||
const fix = bestFuzzyMatch(term, buckets);
|
||||
|
||||
if(fix && fix !== term.toLowerCase()) {
|
||||
changed = true;
|
||||
return fix;
|
||||
}
|
||||
|
||||
return term;
|
||||
});
|
||||
|
||||
return changed ? corrected : null;
|
||||
}
|
||||
+202
-56
@@ -6,13 +6,16 @@ import net from 'node:net';
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import {fileURLToPath} from 'node:url';
|
||||
import {fromXml} from '@ztimson/utils';
|
||||
import {fromXml, makeArray} from '@ztimson/utils';
|
||||
import {titleVocabulary} from './reader.js';
|
||||
import {diversify, rerank, rrfMerge, suggestCorrection} from './search.js';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin'); // npm package root/bin - where bin/install.js drops the kiwix-tools binaries
|
||||
const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin');
|
||||
const READY_TIMEOUT = 10_000;
|
||||
const READY_POLL_INTERVAL = 100;
|
||||
const WATCH_DEBOUNCE = 300;
|
||||
|
||||
function findFreePort() {
|
||||
return new Promise((resolve, reject) => {
|
||||
@@ -25,23 +28,27 @@ function findFreePort() {
|
||||
});
|
||||
}
|
||||
|
||||
/** Owns a kiwix-serve process's full lifecycle: library.xml, start/stop/reload, content + search access. */
|
||||
function tokenize(terms) {
|
||||
return String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
|
||||
}
|
||||
|
||||
export class KiwixServer {
|
||||
static #empty = '<?xml version="1.0" encoding="UTF-8" ?>\n<library version="20110515"></library>\n';
|
||||
|
||||
#dir;
|
||||
#host;
|
||||
#port;
|
||||
#binDir;
|
||||
#libraryPath;
|
||||
#child = null;
|
||||
#remote; // baseUrl string if attached to an externally-managed kiwix-serve, else null
|
||||
#remote;
|
||||
#watcher = null;
|
||||
#watchTimer = null;
|
||||
|
||||
get port() { return this.#port; }
|
||||
get running() { return !!this.#remote || !!this.#child; }
|
||||
get baseUrl() { return this.#remote || (this.#child ? `http://${this.#host}:${this.#port}` : null); }
|
||||
|
||||
/** @param {{port?: number, host?: string, binDir?: string, url?: string}} [opts]
|
||||
* url: attach to an already-running kiwix-serve (e.g. one started elsewhere in your codebase) instead of
|
||||
* spawning/owning one - start/stop/reload become no-ops, and library.xml is read over HTTP instead of disk. */
|
||||
constructor(dir, {port, host = '127.0.0.1', binDir = DEFAULT_BIN_DIR, url} = {}) {
|
||||
this.#dir = dir;
|
||||
this.#host = host;
|
||||
@@ -49,71 +56,139 @@ export class KiwixServer {
|
||||
this.#binDir = binDir;
|
||||
this.#libraryPath = path.join(dir, 'library.xml');
|
||||
this.#remote = url ? url.replace(/\/$/, '') : null;
|
||||
|
||||
if(!this.#remote) this.#ensureLocalStore();
|
||||
}
|
||||
|
||||
#ensureLocalStore() {
|
||||
fs.mkdirSync(this.#dir, {recursive: true});
|
||||
|
||||
if(!fs.existsSync(this.#libraryPath))
|
||||
fs.writeFileSync(this.#libraryPath, KiwixServer.#empty);
|
||||
}
|
||||
|
||||
#assertRunning() {
|
||||
if (!this.running) throw new Error('KiwixServer is not running - call start() first');
|
||||
if(!this.running) throw new Error('KiwixServer is not running - call start() first');
|
||||
}
|
||||
|
||||
#bin(name) {
|
||||
return path.join(this.#binDir, process.platform === 'win32' ? `${name}.exe` : name);
|
||||
}
|
||||
|
||||
/** Reads library.xml from disk if we own the server, or over HTTP if attached to a remote one. */
|
||||
async #fetchLibraryXml() {
|
||||
if (this.#remote) return (await fetch(`${this.#remote}/library.xml`).catch(() => null))?.text?.() ?? '';
|
||||
if(!this.#remote)
|
||||
return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => '');
|
||||
|
||||
try {
|
||||
const res = await fetch(`${this.#remote}/library.xml`);
|
||||
return res.ok ? await res.text() : '';
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
/** Rebuilds library.xml from scratch by scanning `dir` for .zim files - no-op if attached to a remote server. */
|
||||
async #rebuildLibrary() {
|
||||
if (this.#remote) return;
|
||||
if(this.#remote) return;
|
||||
|
||||
await fs.promises.rm(this.#libraryPath, {force: true});
|
||||
for (const f of await this.#zimFiles()) await execFileAsync(this.#bin('kiwix-manage'), [this.#libraryPath, 'add', path.join(this.#dir, f)]);
|
||||
|
||||
const files = await this.#zimFiles();
|
||||
|
||||
if(!files.length)
|
||||
return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty);
|
||||
|
||||
for(const file of files)
|
||||
await execFileAsync(this.#bin('kiwix-manage'), [
|
||||
this.#libraryPath,
|
||||
'add',
|
||||
path.join(this.#dir, file),
|
||||
]);
|
||||
}
|
||||
|
||||
async #waitUntilReady() {
|
||||
const deadline = Date.now() + READY_TIMEOUT;
|
||||
while (Date.now() < deadline) {
|
||||
|
||||
while(Date.now() < deadline) {
|
||||
try {
|
||||
await fetch(`http://${this.#host}:${this.#port}/`);
|
||||
return;
|
||||
} catch {}
|
||||
|
||||
await new Promise(r => setTimeout(r, READY_POLL_INTERVAL));
|
||||
}
|
||||
|
||||
throw new Error('kiwix-serve did not become ready in time');
|
||||
}
|
||||
|
||||
async #zimFiles() {
|
||||
return (await fs.promises.readdir(this.#dir).catch(() => [])).filter(f => f.endsWith('.zim'));
|
||||
return (await fs.promises.readdir(this.#dir).catch(() => []))
|
||||
.filter(f => f.endsWith('.zim'));
|
||||
}
|
||||
|
||||
#watchDir() {
|
||||
this.#watcher?.close();
|
||||
|
||||
this.#watcher = fs.watch(this.#dir, (_event, filename) => {
|
||||
if(!filename?.endsWith('.zim')) return;
|
||||
|
||||
clearTimeout(this.#watchTimer);
|
||||
this.#watchTimer = setTimeout(() => this.#rebuildLibrary().catch(() => {}), WATCH_DEBOUNCE);
|
||||
});
|
||||
|
||||
this.#watcher.on('error', () => {});
|
||||
}
|
||||
|
||||
#unwatchDir() {
|
||||
clearTimeout(this.#watchTimer);
|
||||
this.#watchTimer = null;
|
||||
this.#watcher?.close();
|
||||
this.#watcher = null;
|
||||
}
|
||||
|
||||
/** Rebuilds library.xml and starts kiwix-serve. Resolves once the server is responding. */
|
||||
async start() {
|
||||
if (this.#remote || this.#child) return;
|
||||
if(this.#remote || this.#child) return;
|
||||
|
||||
await fs.promises.mkdir(this.#dir, {recursive: true});
|
||||
await this.#rebuildLibrary();
|
||||
this.#port ??= await findFreePort();
|
||||
|
||||
this.#child = spawn(this.#bin('kiwix-serve'), ['--library', '-i', this.#host, '-p', String(this.#port), this.#libraryPath], {stdio: 'ignore'});
|
||||
this.#child.on('exit', () => { this.#child = null; });
|
||||
this.#child = spawn(this.#bin('kiwix-serve'), [
|
||||
'--library',
|
||||
'--monitorLibrary',
|
||||
'-i',
|
||||
this.#host,
|
||||
'-p',
|
||||
String(this.#port),
|
||||
this.#libraryPath,
|
||||
], {stdio: 'ignore'});
|
||||
|
||||
this.#child.on('exit', () => {
|
||||
this.#child = null;
|
||||
this.#unwatchDir();
|
||||
});
|
||||
|
||||
try {
|
||||
await this.#waitUntilReady();
|
||||
} catch (e) {
|
||||
} catch(e) {
|
||||
await this.stop();
|
||||
throw e;
|
||||
}
|
||||
|
||||
this.#watchDir();
|
||||
}
|
||||
|
||||
/** Gracefully stops kiwix-serve, if we own it. No-op if attached to a remote instance. */
|
||||
async stop() {
|
||||
if (this.#remote || !this.#child) return;
|
||||
this.#unwatchDir();
|
||||
|
||||
if(this.#remote || !this.#child) return;
|
||||
|
||||
const child = this.#child;
|
||||
|
||||
await new Promise(resolve => {
|
||||
child.once('exit', resolve);
|
||||
child.kill('SIGTERM');
|
||||
});
|
||||
|
||||
this.#child = null;
|
||||
}
|
||||
|
||||
@@ -122,21 +197,23 @@ export class KiwixServer {
|
||||
await this.start();
|
||||
}
|
||||
|
||||
/** Rebuilds library.xml from disk and restarts kiwix-serve. No-op if attached to a remote instance -
|
||||
* whoever owns that process is responsible for reloading it. */
|
||||
async reload() {
|
||||
if (this.#remote || !this.#child) return;
|
||||
await this.restart();
|
||||
if(this.#remote || !this.#child) return;
|
||||
await this.#rebuildLibrary();
|
||||
}
|
||||
|
||||
/** Local catalog listing - same flat shape as the online catalog (catalog.js), plus a `file` field. */
|
||||
async list() {
|
||||
this.#assertRunning();
|
||||
|
||||
const xml = await this.#fetchLibraryXml();
|
||||
if(!xml) return [];
|
||||
|
||||
const entries = fromXml(xml);
|
||||
return (entries?.library?.book || []).map(e => {
|
||||
|
||||
return makeArray(entries?.library?.book || []).map(e => {
|
||||
const tags = e.tags.split(';');
|
||||
const name = e.path.replaceAll('.zim', '');
|
||||
const name = e.path.replace(/\.zim$/i, '');
|
||||
|
||||
return {
|
||||
id: e.id,
|
||||
title: e.title,
|
||||
@@ -151,6 +228,7 @@ export class KiwixServer {
|
||||
publisher: e.publisher,
|
||||
articleCount: +e.articleCount || 0,
|
||||
sizeMb: +(Number(e.size) / 1024).toFixed(1) || 0,
|
||||
file: e.path,
|
||||
href: name,
|
||||
icon: `data:${e.faviconMimetype || 'image/png'};base64,${e.favicon}`,
|
||||
viewer: `${this.baseUrl}/content/${name}`,
|
||||
@@ -158,58 +236,126 @@ export class KiwixServer {
|
||||
});
|
||||
}
|
||||
|
||||
/** Splits a href ("zim/path/to/page") or a full content/viewer URL into {zim, path}. */
|
||||
#splitHref(href) {
|
||||
const clean = href.replace(`${this.baseUrl}/content/`, '').replace(/^\/+/, '');
|
||||
const [zim, ...rest] = clean.split('/');
|
||||
return {zim, path: rest.join('/')};
|
||||
}
|
||||
|
||||
/** Builds a kiwix-serve content URL from a href (as returned by list()/search()), or from an already-built content/viewer URL. */
|
||||
link(href) {
|
||||
this.#assertRunning();
|
||||
|
||||
const {zim, path} = this.#splitHref(href);
|
||||
|
||||
return `${this.baseUrl}/content/${zim}${path ? '/' + path : ''}`;
|
||||
}
|
||||
|
||||
/** Fetches a single asset's raw bytes straight from kiwix-serve. */
|
||||
async raw(href) {
|
||||
const res = await fetch(this.link(href));
|
||||
|
||||
if(!res.ok) return null;
|
||||
return {mimetype: res.headers.get('content-type'), data: Buffer.from(await res.arrayBuffer())};
|
||||
|
||||
return {
|
||||
mimetype: res.headers.get('content-type'),
|
||||
data: Buffer.from(await res.arrayBuffer()),
|
||||
};
|
||||
}
|
||||
|
||||
/** Fulltext search across every local ZIM via kiwix-serve's own xapian index */
|
||||
async search(terms, limit = 20) {
|
||||
this.#assertRunning();
|
||||
const termList = String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
|
||||
if (!termList.length) return [];
|
||||
async #rawSearch(termList, scoped, bookMap, limit) {
|
||||
const perBook = await Promise.all(scoped.map(async book => {
|
||||
const params = new URLSearchParams({
|
||||
pattern: termList.join(' '),
|
||||
format: 'xml',
|
||||
pageLength: String(limit),
|
||||
'books.name': book.href,
|
||||
});
|
||||
|
||||
const params = new URLSearchParams({pattern: termList.join(' '), format: 'xml', pageLength: String(limit)});
|
||||
const res = await fetch(`${this.baseUrl}/search?${params}`);
|
||||
if (!res.ok) return [];
|
||||
if(!res.ok) return [];
|
||||
|
||||
const found = fromXml(await res.text())?.rss?.channel?.item || [];
|
||||
|
||||
return found.map(hit => {
|
||||
const resultBook = bookMap.get(hit.book?.title) || book;
|
||||
|
||||
const prefix = `/content/${resultBook.href}/`;
|
||||
const page = hit.link.startsWith(prefix)
|
||||
? hit.link.slice(prefix.length)
|
||||
: hit.link.replace(/^\/+/, '');
|
||||
|
||||
return {
|
||||
id: resultBook.id,
|
||||
title: hit.title,
|
||||
page,
|
||||
name: resultBook.name,
|
||||
publisher: resultBook.publisher,
|
||||
href: `${resultBook.href}/${page}`,
|
||||
icon: resultBook.icon,
|
||||
viewer: this.baseUrl + hit.link,
|
||||
summary: hit.description,
|
||||
xapianScore: +hit.score || 0,
|
||||
};
|
||||
}).filter(Boolean);
|
||||
}));
|
||||
|
||||
return rrfMerge(perBook);
|
||||
}
|
||||
|
||||
/** Fulltext search across local ZIMs with ranking, diversification and spelling correction. */
|
||||
async search(terms, {limit = 20, sources = null} = {}) {
|
||||
this.#assertRunning();
|
||||
|
||||
const termList = tokenize(terms);
|
||||
if(!termList.length) return {results: [], spellcheck: null};
|
||||
|
||||
const books = await this.list();
|
||||
const bookMap = new Map(books.map(b => [b.title, b]));
|
||||
|
||||
return found.map(hit => {
|
||||
const book = bookMap.get(hit.book.title);
|
||||
if (!book) return null;
|
||||
const prefix = `/content/${book.href}/`;
|
||||
const page = hit.link.startsWith(prefix) ? hit.link.slice(prefix.length) : hit.link.replace(/^\/+/, '');
|
||||
const scoped = sources?.length
|
||||
? books.filter(b => sources.includes(b.name) || sources.includes(b.href))
|
||||
: books;
|
||||
|
||||
if(!scoped.length) return {results: [], spellcheck: null};
|
||||
|
||||
let activeTerms = termList;
|
||||
let hits = await this.#rawSearch(activeTerms, scoped, bookMap, limit);
|
||||
let spellcheck = null;
|
||||
|
||||
if(!hits.length && !this.#remote) {
|
||||
const vocabulary = new Set();
|
||||
|
||||
for(const book of scoped) {
|
||||
if(!book.file) continue;
|
||||
|
||||
try {
|
||||
for(const word of await titleVocabulary(path.join(this.#dir, book.file)))
|
||||
vocabulary.add(word);
|
||||
} catch {
|
||||
// Unreadable ZIM - skip it, don't fail the whole search.
|
||||
}
|
||||
}
|
||||
|
||||
const corrected = suggestCorrection(termList, vocabulary);
|
||||
|
||||
if(corrected) {
|
||||
const retry = await this.#rawSearch(corrected, scoped, bookMap, limit);
|
||||
|
||||
if(retry.length) {
|
||||
hits = retry;
|
||||
activeTerms = corrected;
|
||||
spellcheck = {
|
||||
from: termList.join(' '),
|
||||
to: corrected.join(' '),
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const ranked = rerank(hits, activeTerms);
|
||||
|
||||
return {
|
||||
id: book.id,
|
||||
title: hit.title,
|
||||
page,
|
||||
name: book.name,
|
||||
publisher: book.publisher,
|
||||
href: `${book.href}/${page}`,
|
||||
icon: book.icon,
|
||||
viewer: this.baseUrl + hit.link,
|
||||
summary: hit.description,
|
||||
score: +hit.score || 0,
|
||||
results: diversify(ranked, limit),
|
||||
spellcheck,
|
||||
};
|
||||
}).filter(Boolean);
|
||||
}
|
||||
}
|
||||
|
||||
+12
-15
@@ -1,39 +1,36 @@
|
||||
export function levenshtein(a, b) {
|
||||
const m = a.length, n = b.length;
|
||||
if (!m) return n;
|
||||
if (!n) return m;
|
||||
if(!m) return n;
|
||||
if(!n) return m;
|
||||
const dp = Array.from({length: m + 1}, (_, i) => [i, ...Array(n).fill(0)]);
|
||||
for (let j = 0; j <= n; j++) dp[0][j] = j;
|
||||
for (let i = 1; i <= m; i++) {
|
||||
for (let j = 1; j <= n; j++) {
|
||||
dp[i][j] = a[i - 1] === b[j - 1]
|
||||
? dp[i - 1][j - 1]
|
||||
: 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
|
||||
for(let j = 0; j <= n; j++) dp[0][j] = j;
|
||||
for(let i = 1; i <= m; i++) {
|
||||
for(let j = 1; j <= n; j++) {
|
||||
dp[i][j] = a[i - 1] === b[j - 1] ? dp[i - 1][j - 1] : 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
|
||||
}
|
||||
}
|
||||
return dp[m][n];
|
||||
}
|
||||
|
||||
function scoreAgainst(text, term) {
|
||||
if (text.includes(term)) return 1 - (text.length - term.length) / text.length * 0.3;
|
||||
if (!text.length || !term.length || text[0] !== term[0]) return 0;
|
||||
const dist = levenshtein(text, term);
|
||||
if (!text.length || !term.length) return 0;
|
||||
if (text === term) return 1;
|
||||
if (text.includes(term)) return 0.8;
|
||||
const maxAllowed = Math.max(1, Math.ceil(term.length * 0.34));
|
||||
if (Math.abs(text.length - term.length) > maxAllowed) return 0;
|
||||
const dist = levenshtein(text, term);
|
||||
if (dist > maxAllowed) return 0;
|
||||
return 1 - dist / Math.max(text.length, term.length);
|
||||
}
|
||||
|
||||
/** Compares `target` against one or more search terms; returns avg/max/per-term similarity. */
|
||||
export function fuzzyMatch(target, ...terms) {
|
||||
if (!terms.length) throw new Error('Requires at least 1 term to compare');
|
||||
const lowerTarget = String(target).toLowerCase();
|
||||
const words = lowerTarget.split(/\W+/).filter(Boolean);
|
||||
|
||||
const similarities = terms.map(term => {
|
||||
const t = term.toLowerCase();
|
||||
const t = String(term).toLowerCase();
|
||||
return Math.max(scoreAgainst(lowerTarget, t), ...words.map(w => scoreAgainst(w, t)));
|
||||
});
|
||||
|
||||
return {
|
||||
avg: similarities.reduce((acc, s) => acc + s, 0) / similarities.length,
|
||||
max: Math.max(...similarities),
|
||||
|
||||
Reference in New Issue
Block a user