generated from ztimson/template
Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 7d376c90b4 | |||
| bfb3f0efd3 | |||
| 4ef022aa0e | |||
| fb7ae55d49 | |||
| bc90557de7 |
10
.gitignore
vendored
Normal file
10
.gitignore
vendored
Normal file
@@ -0,0 +1,10 @@
|
|||||||
|
.idea
|
||||||
|
.vscode
|
||||||
|
|
||||||
|
logs
|
||||||
|
*.log
|
||||||
|
|
||||||
|
bin/*.dll
|
||||||
|
bin/kiwix*
|
||||||
|
node_modles
|
||||||
|
zims
|
||||||
87
bin/install-kwix.js
Normal file
87
bin/install-kwix.js
Normal file
@@ -0,0 +1,87 @@
|
|||||||
|
import fs from 'node:fs';
|
||||||
|
import os from 'node:os';
|
||||||
|
import path from 'node:path';
|
||||||
|
import https from 'node:https';
|
||||||
|
import {execFileSync} from 'node:child_process';
|
||||||
|
import {fileURLToPath} from 'node:url';
|
||||||
|
|
||||||
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||||
|
|
||||||
|
const VERSION = '3.8.1';
|
||||||
|
const BASE_URL = 'https://download.kiwix.org/release/kiwix-tools';
|
||||||
|
const PLATFORM_MAP = {linux: 'linux', darwin: 'macos', win32: 'win'};
|
||||||
|
|
||||||
|
const BIN_NAME = process.platform === 'win32' ? 'kiwix-search.exe' : 'kiwix-search';
|
||||||
|
const BIN_DIR = path.join(__dirname, '..', 'bin'); // project root/bin
|
||||||
|
const BIN_PATH = path.join(BIN_DIR, BIN_NAME);
|
||||||
|
|
||||||
|
/** Download a file, following redirects. */
|
||||||
|
function download(url, dest) {
|
||||||
|
return new Promise((resolve, reject) => {
|
||||||
|
const file = fs.createWriteStream(dest);
|
||||||
|
https.get(url, res => {
|
||||||
|
if (res.statusCode >= 300 && res.statusCode < 400 && res.headers.location) {
|
||||||
|
file.close();
|
||||||
|
return resolve(download(res.headers.location, dest));
|
||||||
|
}
|
||||||
|
if (!res.statusCode || res.statusCode >= 400) {
|
||||||
|
return reject(new Error(`Download failed: ${res.statusCode} ${res.statusMessage}`));
|
||||||
|
}
|
||||||
|
res.pipe(file);
|
||||||
|
file.on('finish', () => file.close(resolve));
|
||||||
|
}).on('error', err => fs.unlink(dest, () => reject(err)));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Recursively search a directory for a file by name. */
|
||||||
|
function findFile(dir, name) {
|
||||||
|
for (const entry of fs.readdirSync(dir, {withFileTypes: true})) {
|
||||||
|
const full = path.join(dir, entry.name);
|
||||||
|
if (entry.isDirectory()) {
|
||||||
|
const found = findFile(full, name);
|
||||||
|
if (found) return found;
|
||||||
|
} else if (entry.name === name) {
|
||||||
|
return full;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
async function downloadAndExtract() {
|
||||||
|
const platformName = PLATFORM_MAP[process.platform];
|
||||||
|
if (!platformName) throw new Error(`Unsupported platform: ${process.platform}`);
|
||||||
|
|
||||||
|
const ext = platformName === 'win' ? 'zip' : 'tar.gz';
|
||||||
|
const archiveName = `kiwix-tools_${platformName}-x86_64-${VERSION}.${ext}`;
|
||||||
|
const url = `${BASE_URL}/${archiveName}`;
|
||||||
|
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'kiwix-tools-'));
|
||||||
|
const archivePath = path.join(tmpDir, archiveName);
|
||||||
|
|
||||||
|
console.log('Downloading kiwix-tools from:', url);
|
||||||
|
await download(url, archivePath);
|
||||||
|
|
||||||
|
console.log('Extracting...');
|
||||||
|
execFileSync('tar', ['-xf', archivePath, '-C', tmpDir]); // bsdtar handles zip too
|
||||||
|
|
||||||
|
const extractedBin = findFile(tmpDir, BIN_NAME);
|
||||||
|
if (!extractedBin) throw new Error(`Could not find ${BIN_NAME} in extracted archive`);
|
||||||
|
const extractedDir = path.dirname(extractedBin);
|
||||||
|
|
||||||
|
await fs.promises.mkdir(BIN_DIR, {recursive: true});
|
||||||
|
|
||||||
|
// Copy the binary plus any DLLs sitting alongside it (Windows deps)
|
||||||
|
for (const entry of await fs.promises.readdir(extractedDir)) {
|
||||||
|
if (entry === BIN_NAME || entry.toLowerCase().endsWith('.dll')) {
|
||||||
|
await fs.promises.copyFile(path.join(extractedDir, entry), path.join(BIN_DIR, entry));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (process.platform !== 'win32') await fs.promises.chmod(BIN_PATH, 0o755);
|
||||||
|
|
||||||
|
await fs.promises.rm(tmpDir, {recursive: true, force: true});
|
||||||
|
console.log('Installed to:', BIN_PATH);
|
||||||
|
}
|
||||||
|
|
||||||
|
downloadAndExtract().catch(err => {
|
||||||
|
console.error(err);
|
||||||
|
process.exit(1)
|
||||||
|
}).finally(() => process.exit());
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@ztimson/zim-utils",
|
"name": "@ztimson/zim-utils",
|
||||||
"version": "0.2.3",
|
"version": "0.2.5",
|
||||||
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
|
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
|
||||||
"author": "Zak Timson",
|
"author": "Zak Timson",
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
@@ -11,6 +11,9 @@
|
|||||||
"url": "https://git.zakscode.com/ztimson/zim-utils"
|
"url": "https://git.zakscode.com/ztimson/zim-utils"
|
||||||
},
|
},
|
||||||
"main": "src/index.js",
|
"main": "src/index.js",
|
||||||
|
"scripts": {
|
||||||
|
"postinstall": "node ./bin/install-kwix.js"
|
||||||
|
},
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@ztimson/utils": "^0.30.7",
|
"@ztimson/utils": "^0.30.7",
|
||||||
"lzma1": "^0.3.0",
|
"lzma1": "^0.3.0",
|
||||||
|
|||||||
303
src/reader.js
303
src/reader.js
@@ -1,11 +1,10 @@
|
|||||||
'use strict';
|
'use strict';
|
||||||
|
|
||||||
import fs from 'node:fs';
|
import fs from 'node:fs';
|
||||||
import path from 'node:path';
|
import {createHash} from 'node:crypto';
|
||||||
import {decompressPool} from './decompress.js';
|
import {decompressPool} from './decompress.js';
|
||||||
import {fuzzyMatch, titleFromUrl} from './utils.js';
|
import {fuzzyMatch, titleFromUrl, kiwixSearch} from './utils.js';
|
||||||
|
|
||||||
const INDEX_VERSION = 1;
|
|
||||||
const HEADER_SIZE = 80;
|
const HEADER_SIZE = 80;
|
||||||
const NS_CONTENT = 'C';
|
const NS_CONTENT = 'C';
|
||||||
const NS_METADATA = 'M';
|
const NS_METADATA = 'M';
|
||||||
@@ -14,18 +13,29 @@ const TITLE_SENTINEL = 0xffffffffffffffffn; // Indicator -> ZIM v6+ archives wit
|
|||||||
const DEFAULT_CLUSTER_CACHE_MAX = 32;
|
const DEFAULT_CLUSTER_CACHE_MAX = 32;
|
||||||
const DEFAULT_CLUSTER_TTL = 60_000;
|
const DEFAULT_CLUSTER_TTL = 60_000;
|
||||||
|
|
||||||
|
// --- Fallback search index (only used for archives with no native title listing) ---
|
||||||
|
const INDEX_SUFFIX = '.searchidx.bin';
|
||||||
|
const INDEX_MAGIC = 'ZXI1';
|
||||||
|
const INDEX_HEADER_SIZE = 24; // magic(4) + staleness key(16) + recordCount(4)
|
||||||
|
const INDEX_RECORD_SIZE = 20; // keyOff(4) keyLen(2) urlOff(4) urlLen(2) titleOff(4) titleLen(2) mimetype(2)
|
||||||
|
|
||||||
/** Native, dependency-light reader for .zim archives. Supports zstd & LZMA cluster compression. */
|
/** Native, dependency-light reader for .zim archives. Supports zstd & LZMA cluster compression. */
|
||||||
export class ZimReader {
|
export class ZimReader {
|
||||||
#fd = null;
|
#fd = null;
|
||||||
#header = null;
|
#header = null;
|
||||||
#mimeTypes = [];
|
#mimeTypes = [];
|
||||||
#hasTitleListing = false;
|
#hasTitleListing = false;
|
||||||
#index;
|
|
||||||
#clusterCache = new Map(); // clusterNumber -> {data, extended, timer}
|
#clusterCache = new Map(); // clusterNumber -> {data, extended, timer}
|
||||||
#pending = new Map(); // clusterNumber -> Promise, dedupes concurrent misses
|
#pending = new Map(); // clusterNumber -> Promise, dedupes concurrent misses
|
||||||
#clusterCacheMax;
|
#clusterCacheMax;
|
||||||
#clusterTTL;
|
#clusterTTL;
|
||||||
|
|
||||||
|
#indexPath;
|
||||||
|
#indexReady = null; // Promise, awaited by search() before using the fallback index
|
||||||
|
#indexFd = null; // open fd for the fallback index, once loaded/built
|
||||||
|
#indexRecordCount = 0;
|
||||||
|
#indexTableStart = 0;
|
||||||
|
|
||||||
get articleCount() { return this.#header?.articleCount ?? 0; }
|
get articleCount() { return this.#header?.articleCount ?? 0; }
|
||||||
get mediaCount() { return this.#header?.clusterCount ?? 0; }
|
get mediaCount() { return this.#header?.clusterCount ?? 0; }
|
||||||
|
|
||||||
@@ -34,18 +44,10 @@ export class ZimReader {
|
|||||||
this.path = path;
|
this.path = path;
|
||||||
this.#clusterCacheMax = clusterCacheMax;
|
this.#clusterCacheMax = clusterCacheMax;
|
||||||
this.#clusterTTL = clusterTTL;
|
this.#clusterTTL = clusterTTL;
|
||||||
|
this.#indexPath = `${path}${INDEX_SUFFIX}`;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Full O(n) scan over the URL pointer list, used when there's no title index. */
|
/** Binary search the URL pointer list for namespace+url. For exact-key lookups (readPage, metadata, icons). */
|
||||||
async #allDirents() {
|
|
||||||
const dirents = [];
|
|
||||||
for (let i = 0; i < this.#header.articleCount; i++) {
|
|
||||||
dirents.push(await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, i)));
|
|
||||||
}
|
|
||||||
return dirents;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Binary search the URL pointer list for namespace+url. */
|
|
||||||
async #findByUrl(url, namespace) {
|
async #findByUrl(url, namespace) {
|
||||||
const key = namespace + url;
|
const key = namespace + url;
|
||||||
let lo = 0, hi = this.#header.articleCount - 1;
|
let lo = 0, hi = this.#header.articleCount - 1;
|
||||||
@@ -60,6 +62,182 @@ export class ZimReader {
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Binary search the ZIM's own title pointer list. O(log n), zero extra storage - the happy path. */
|
||||||
|
async #findByTitleBuiltin(title) {
|
||||||
|
let lo = 0, hi = this.#header.articleCount - 1;
|
||||||
|
while (lo <= hi) {
|
||||||
|
const mid = (lo + hi) >> 1;
|
||||||
|
const urlIdx = (await this.#read(this.#header.titlePtrPos + mid * 4, 4)).readUInt32LE(0);
|
||||||
|
const d = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx));
|
||||||
|
if (d.title === title) return d;
|
||||||
|
if (d.title < title) lo = mid + 1; else hi = mid - 1;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Resolves a kiwix-search result (a title, not necessarily a url) to
|
||||||
|
* {url, title, mimetype}. kiwix-search's fulltext index returns titles, and
|
||||||
|
* a title isn't guaranteed to equal its url (unicode normalization,
|
||||||
|
* disambiguation suffixes, punctuation stripping), so:
|
||||||
|
* 1. URL binary search - matches when title happens to equal url (free to check)
|
||||||
|
* 2. Title pointer list - when the archive ships one (Wikipedia etc. do)
|
||||||
|
* 3. Persisted fallback index / linear scan - only for archives without (2)
|
||||||
|
*/
|
||||||
|
async #resolveSearchEntry(name) {
|
||||||
|
let dirent = await this.#findByUrl(name, NS_CONTENT);
|
||||||
|
if (dirent) return dirent;
|
||||||
|
|
||||||
|
if (this.#hasTitleListing) return this.#findByTitleBuiltin(name);
|
||||||
|
|
||||||
|
if (this.#indexReady) await this.#indexReady;
|
||||||
|
const hit = await this.#lookupFallbackIndex(name);
|
||||||
|
if (hit) return hit;
|
||||||
|
|
||||||
|
// Index unavailable (build failed - unwritable disk, etc.) or genuinely no match.
|
||||||
|
for (let i = 0; i < this.#header.articleCount; i++) {
|
||||||
|
const d = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, i));
|
||||||
|
if (d.namespace === NS_CONTENT && (d.title === name || titleFromUrl(d.url) === name)) return d;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A cheap, stable fingerprint for "is the cached index still valid for this file".
|
||||||
|
* ZIM archives end with a 16-byte MD5 of their own contents (the same trailer
|
||||||
|
* `zimcheck` validates against), so this is a single 16-byte read regardless of
|
||||||
|
* archive size - no need to hash a multi-hundred-GB file. It's also
|
||||||
|
* content-based rather than path/mtime-based, so moving or redownloading an
|
||||||
|
* identical archive doesn't invalidate the cache. Falls back to a tiny
|
||||||
|
* size+mtime hash only if the file is too short to have a real trailer.
|
||||||
|
*/
|
||||||
|
async #stalenessKey() {
|
||||||
|
const {size} = await fs.promises.stat(this.path);
|
||||||
|
if (size >= 16) return this.#read(size - 16, 16);
|
||||||
|
const stat = await fs.promises.stat(this.path);
|
||||||
|
return createHash('md5').update(`${stat.size}:${stat.mtimeMs}`).digest();
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Walks every content dirent once, resolving redirects, keyed by both its title and its url-derived title. */
|
||||||
|
async #collectFallbackEntries() {
|
||||||
|
const map = new Map(); // key -> {url, title, mimetype}
|
||||||
|
for (let i = 0; i < this.#header.articleCount; i++) {
|
||||||
|
let dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, i));
|
||||||
|
if (dirent.namespace !== NS_CONTENT) continue;
|
||||||
|
dirent = await this.#resolveRedirect(dirent);
|
||||||
|
if (dirent.namespace !== NS_CONTENT) continue; // redirected outside content namespace
|
||||||
|
|
||||||
|
const entry = {url: dirent.url, title: dirent.title, mimetype: dirent.mimetype};
|
||||||
|
if (!map.has(dirent.title)) map.set(dirent.title, entry);
|
||||||
|
const urlTitle = titleFromUrl(dirent.url);
|
||||||
|
if (urlTitle !== dirent.title && !map.has(urlTitle)) map.set(urlTitle, entry);
|
||||||
|
}
|
||||||
|
return map;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Serializes the fallback index to disk: header, sorted fixed-width records, then a string table. */
|
||||||
|
async #buildFallbackIndex(stalenessKey) {
|
||||||
|
const map = await this.#collectFallbackEntries();
|
||||||
|
const keys = [...map.keys()].sort();
|
||||||
|
|
||||||
|
const records = Buffer.alloc(keys.length * INDEX_RECORD_SIZE);
|
||||||
|
const strings = [];
|
||||||
|
let tableOffset = 0;
|
||||||
|
keys.forEach((key, i) => {
|
||||||
|
const {url, title, mimetype} = map.get(key);
|
||||||
|
const keyBuf = Buffer.from(key, 'utf8');
|
||||||
|
const urlBuf = Buffer.from(url, 'utf8');
|
||||||
|
const titleBuf = Buffer.from(title, 'utf8');
|
||||||
|
const base = i * INDEX_RECORD_SIZE;
|
||||||
|
|
||||||
|
records.writeUInt32LE(tableOffset, base); records.writeUInt16LE(keyBuf.length, base + 4);
|
||||||
|
tableOffset += keyBuf.length;
|
||||||
|
records.writeUInt32LE(tableOffset, base + 6); records.writeUInt16LE(urlBuf.length, base + 10);
|
||||||
|
tableOffset += urlBuf.length;
|
||||||
|
records.writeUInt32LE(tableOffset, base + 12); records.writeUInt16LE(titleBuf.length, base + 16);
|
||||||
|
tableOffset += titleBuf.length;
|
||||||
|
records.writeUInt16LE(mimetype, base + 18);
|
||||||
|
|
||||||
|
strings.push(keyBuf, urlBuf, titleBuf);
|
||||||
|
});
|
||||||
|
|
||||||
|
const header = Buffer.alloc(INDEX_HEADER_SIZE);
|
||||||
|
header.write(INDEX_MAGIC, 0, 'ascii');
|
||||||
|
stalenessKey.copy(header, 4);
|
||||||
|
header.writeUInt32LE(keys.length, 20);
|
||||||
|
|
||||||
|
const tmpPath = `${this.#indexPath}.tmp-${process.pid}`;
|
||||||
|
await fs.promises.writeFile(tmpPath, Buffer.concat([header, records, ...strings]));
|
||||||
|
await fs.promises.rename(tmpPath, this.#indexPath);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Opens the fallback index file and caches its header fields for querying. */
|
||||||
|
async #openFallbackIndex() {
|
||||||
|
const fd = await fs.promises.open(this.#indexPath, 'r');
|
||||||
|
const header = Buffer.alloc(INDEX_HEADER_SIZE);
|
||||||
|
await fd.read(header, 0, INDEX_HEADER_SIZE, 0);
|
||||||
|
this.#indexFd = fd;
|
||||||
|
this.#indexRecordCount = header.readUInt32LE(20);
|
||||||
|
this.#indexTableStart = INDEX_HEADER_SIZE + this.#indexRecordCount * INDEX_RECORD_SIZE;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Validates the on-disk fallback index against the archive's current staleness
|
||||||
|
* key and (re)builds it if missing/stale. Failures are swallowed - callers see
|
||||||
|
* #indexFd stay null and fall through to the linear-scan last resort.
|
||||||
|
*/
|
||||||
|
async #loadOrBuildFallbackIndex() {
|
||||||
|
const stalenessKey = await this.#stalenessKey();
|
||||||
|
|
||||||
|
try {
|
||||||
|
const fd = await fs.promises.open(this.#indexPath, 'r');
|
||||||
|
const header = Buffer.alloc(INDEX_HEADER_SIZE);
|
||||||
|
await fd.read(header, 0, INDEX_HEADER_SIZE, 0);
|
||||||
|
const fresh = header.toString('ascii', 0, 4) === INDEX_MAGIC && header.subarray(4, 20).equals(stalenessKey);
|
||||||
|
if (fresh) {
|
||||||
|
this.#indexFd = fd;
|
||||||
|
this.#indexRecordCount = header.readUInt32LE(20);
|
||||||
|
this.#indexTableStart = INDEX_HEADER_SIZE + this.#indexRecordCount * INDEX_RECORD_SIZE;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
await fd.close();
|
||||||
|
} catch { /* missing, corrupt, or unreadable -> rebuild below */ }
|
||||||
|
|
||||||
|
await this.#buildFallbackIndex(stalenessKey);
|
||||||
|
await this.#openFallbackIndex();
|
||||||
|
}
|
||||||
|
|
||||||
|
async #indexRead(pos, length) {
|
||||||
|
const buf = Buffer.alloc(length);
|
||||||
|
await this.#indexFd.read(buf, 0, length, pos);
|
||||||
|
return buf;
|
||||||
|
}
|
||||||
|
|
||||||
|
async #indexString(offset, length) {
|
||||||
|
if (!length) return '';
|
||||||
|
return (await this.#indexRead(this.#indexTableStart + offset, length)).toString('utf8');
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Binary search over the on-disk record array. Never loads the full index into memory. */
|
||||||
|
async #lookupFallbackIndex(name) {
|
||||||
|
if (!this.#indexFd) return null;
|
||||||
|
let lo = 0, hi = this.#indexRecordCount - 1;
|
||||||
|
while (lo <= hi) {
|
||||||
|
const mid = (lo + hi) >> 1;
|
||||||
|
const rec = await this.#indexRead(INDEX_HEADER_SIZE + mid * INDEX_RECORD_SIZE, INDEX_RECORD_SIZE);
|
||||||
|
const keyOff = rec.readUInt32LE(0), keyLen = rec.readUInt16LE(4);
|
||||||
|
const key = await this.#indexString(keyOff, keyLen);
|
||||||
|
if (key === name) {
|
||||||
|
const urlOff = rec.readUInt32LE(6), urlLen = rec.readUInt16LE(10);
|
||||||
|
const titleOff = rec.readUInt32LE(12), titleLen = rec.readUInt16LE(16);
|
||||||
|
const [url, title] = await Promise.all([this.#indexString(urlOff, urlLen), this.#indexString(titleOff, titleLen)]);
|
||||||
|
return {url, title, mimetype: rec.readUInt16LE(18)};
|
||||||
|
}
|
||||||
|
if (key < name) lo = mid + 1; else hi = mid - 1;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
/** Resets a cluster's idle-eviction timer. No-op when TTL disabled. */
|
/** Resets a cluster's idle-eviction timer. No-op when TTL disabled. */
|
||||||
#touch(clusterNumber, entry) {
|
#touch(clusterNumber, entry) {
|
||||||
if (!this.#clusterTTL) return;
|
if (!this.#clusterTTL) return;
|
||||||
@@ -185,82 +363,21 @@ export class ZimReader {
|
|||||||
return this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, dirent.redirectIndex));
|
return this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, dirent.redirectIndex));
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Narrows to dirents near the term's alphabetical position in the title index. */
|
|
||||||
async #titleIndexCandidates(term) {
|
|
||||||
const q = term.toLowerCase();
|
|
||||||
let lo = 0, hi = this.#header.articleCount - 1;
|
|
||||||
while (lo < hi) {
|
|
||||||
const mid = (lo + hi) >> 1;
|
|
||||||
const urlIdx = (await this.#read(this.#header.titlePtrPos + mid * 4, 4)).readUInt32LE(0);
|
|
||||||
const dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx));
|
|
||||||
if (dirent.title.toLowerCase() < q) lo = mid + 1; else hi = mid;
|
|
||||||
}
|
|
||||||
// Widen around the prefix match since fuzzy scoring isn't purely alphabetical.
|
|
||||||
const start = Math.max(0, lo - 50), end = Math.min(this.#header.articleCount, lo + 200);
|
|
||||||
const dirents = [];
|
|
||||||
for (let i = start; i < end; i++) {
|
|
||||||
const urlIdx = (await this.#read(this.#header.titlePtrPos + i * 4, 4)).readUInt32LE(0);
|
|
||||||
dirents.push(await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx)));
|
|
||||||
}
|
|
||||||
return dirents;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Path of the cached word index, kept in an `index/` subdirectory next to the archive. */
|
|
||||||
#indexPath() {
|
|
||||||
return path.join(path.dirname(this.path), 'index', `${path.basename(this.path)}.idx.json`);
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Builds (or loads a cached) word -> entry index, avoiding a re-scan on every search. */
|
|
||||||
async #wordIndex() {
|
|
||||||
if (this.#index) return this.#index;
|
|
||||||
const cachePath = this.#indexPath();
|
|
||||||
const stat = await fs.promises.stat(this.path);
|
|
||||||
|
|
||||||
if (fs.existsSync(cachePath)) {
|
|
||||||
try {
|
|
||||||
const cached = JSON.parse(await fs.promises.readFile(cachePath, 'utf8'));
|
|
||||||
// Rebuild if stale: index predates the archive's current mtime (e.g. a re-download).
|
|
||||||
if (cached.version === INDEX_VERSION && cached.mtimeMs >= stat.mtimeMs) {
|
|
||||||
return (this.#index = cached);
|
|
||||||
}
|
|
||||||
} catch {}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Expensive full scan — only happens once per archive (or after it changes).
|
|
||||||
const dirents = await this.#allDirents();
|
|
||||||
const entries = [];
|
|
||||||
const words = new Map(); // word -> [entryIndex, ...]
|
|
||||||
for (const d of dirents) {
|
|
||||||
if (d.namespace !== NS_CONTENT) continue;
|
|
||||||
const idx = entries.length;
|
|
||||||
entries.push({url: d.url, title: d.title, mimetype: d.mimetype});
|
|
||||||
const text = `${d.title} ${titleFromUrl(d.url)}`.toLowerCase();
|
|
||||||
for (const w of text.split(/\W+/).filter(Boolean)) {
|
|
||||||
if (!words.has(w)) words.set(w, []);
|
|
||||||
words.get(w).push(idx);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const index = {version: INDEX_VERSION, size: stat.size, mtimeMs: stat.mtimeMs, entries, words: Object.fromEntries(words)};
|
|
||||||
await fs.promises.mkdir(path.dirname(cachePath), {recursive: true});
|
|
||||||
await fs.promises.writeFile(cachePath, JSON.stringify(index));
|
|
||||||
return (this.#index = index);
|
|
||||||
}
|
|
||||||
|
|
||||||
async close() {
|
async close() {
|
||||||
if (this.#fd) await this.#fd.close();
|
if (this.#fd) await this.#fd.close();
|
||||||
this.#fd = null;
|
this.#fd = null;
|
||||||
|
if (this.#indexFd) await this.#indexFd.close();
|
||||||
|
this.#indexFd = null;
|
||||||
for (const entry of this.#clusterCache.values()) clearTimeout(entry.timer);
|
for (const entry of this.#clusterCache.values()) clearTimeout(entry.timer);
|
||||||
this.#clusterCache.clear();
|
this.#clusterCache.clear();
|
||||||
this.#pending.clear();
|
this.#pending.clear();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Deletes the zim archive and its cached index (if any). Safe to call on unopened readers. */
|
/** Deletes the zim archive and its cached fallback index (if any). Safe to call on unopened readers. */
|
||||||
async delete() {
|
async delete() {
|
||||||
await this.close();
|
await this.close();
|
||||||
this.#index = null;
|
|
||||||
await fs.promises.rm(this.path, {force: true});
|
await fs.promises.rm(this.path, {force: true});
|
||||||
await fs.promises.rm(this.#indexPath(), {force: true});
|
await fs.promises.rm(this.#indexPath, {force: true});
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Reads an 'M' namespace metadata value (e.g. Name, Date, Title). Returns null if missing. */
|
/** Reads an 'M' namespace metadata value (e.g. Name, Date, Title). Returns null if missing. */
|
||||||
@@ -302,6 +419,14 @@ export class ZimReader {
|
|||||||
this.#fd = null;
|
this.#fd = null;
|
||||||
throw e;
|
throw e;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The fallback index is only needed when the archive has no native title
|
||||||
|
// pointer list - #findByTitleBuiltin already covers that case in O(log n)
|
||||||
|
// with zero extra storage. Well-maintained archives (Wikipedia etc.)
|
||||||
|
// ship a title listing, so this path is expected to be rare in practice.
|
||||||
|
if (!this.#hasTitleListing) {
|
||||||
|
this.#indexReady = this.#loadOrBuildFallbackIndex().catch(() => {});
|
||||||
|
}
|
||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -325,27 +450,15 @@ export class ZimReader {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Fuzzy-ranked title search. Accepts comma-separated `terms` the same way the
|
* Fuzzy-ranked title search. Accepts comma-separated `terms` the same way the
|
||||||
* catalog search does. Uses the sorted title index when present (binary search
|
* catalog search does. Uses kiwix-search's embedded fulltext index as a prefilter
|
||||||
* narrows the candidate window); falls back to a full linear scan otherwise
|
* to narrow candidates before fuzzy scoring.
|
||||||
* (common on ZIM v6+/zimit-generated archives with no title index).
|
|
||||||
*/
|
*/
|
||||||
async search(terms, {limit = 20, htmlOnly = true} = {}) {
|
async search(terms, {limit = 20, htmlOnly = true} = {}) {
|
||||||
const termList = String(terms).split(',').map(t => t.trim()).filter(Boolean);
|
const termList = String(terms).split(',').map(t => t.trim()).filter(Boolean);
|
||||||
if (!termList.length) return [];
|
if (!termList.length) return [];
|
||||||
|
|
||||||
let candidates;
|
const names = await kiwixSearch(this.path, termList.join(' '));
|
||||||
if (this.#hasTitleListing && this.#header.articleCount > 5000) {
|
const candidates = (await Promise.all(names.map(n => this.#resolveSearchEntry(n)))).filter(Boolean);
|
||||||
candidates = await this.#titleIndexCandidates(termList[0]);
|
|
||||||
} else {
|
|
||||||
const {entries, words} = await this.#wordIndex();
|
|
||||||
// Pull candidates from postings of any word that starts with (or contains) the search term.
|
|
||||||
const q = termList[0].toLowerCase();
|
|
||||||
const idxSet = new Set();
|
|
||||||
for (const [word, postings] of Object.entries(words)) {
|
|
||||||
if (word.includes(q) || q.includes(word)) postings.forEach(i => idxSet.add(i));
|
|
||||||
}
|
|
||||||
candidates = [...idxSet].map(i => ({...entries[i], namespace: NS_CONTENT}));
|
|
||||||
}
|
|
||||||
|
|
||||||
const scored = [];
|
const scored = [];
|
||||||
for (const dirent of candidates) {
|
for (const dirent of candidates) {
|
||||||
|
|||||||
29
src/utils.js
29
src/utils.js
@@ -1,3 +1,32 @@
|
|||||||
|
import {execFile} from 'node:child_process';
|
||||||
|
import path from 'node:path';
|
||||||
|
import {fileURLToPath} from 'node:url';
|
||||||
|
import {promisify} from 'node:util';
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Search a ZIM file's fulltext index.
|
||||||
|
* @param {string} zimPath - Path to the .zim file
|
||||||
|
* @param {string} pattern - Search terms
|
||||||
|
* @param {object} [opts]
|
||||||
|
* @param {boolean} [opts.suggestion] - Suggest titles from partial pattern (completion-style)
|
||||||
|
* @param {boolean} [opts.spelling] - Suggest spelling-corrected titles
|
||||||
|
* @returns {Promise<string[]>} Matching article/tag titles
|
||||||
|
*/
|
||||||
|
export async function kiwixSearch(zimPath, pattern, opts = {}) {
|
||||||
|
const execFileAsync = promisify(execFile);
|
||||||
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||||
|
const BIN_NAME = process.platform === 'win32' ? 'kiwix-search.exe' : 'kiwix-search';
|
||||||
|
const BIN_PATH = path.join(__dirname, '..', 'bin', BIN_NAME);
|
||||||
|
|
||||||
|
const args = [];
|
||||||
|
if (opts.suggestion) args.push('-s');
|
||||||
|
if (opts.spelling) args.push('--spelling');
|
||||||
|
args.push(zimPath, pattern);
|
||||||
|
|
||||||
|
const {stdout} = await execFileAsync(BIN_PATH, args);
|
||||||
|
return stdout.split('\n').map(line => line.trim()).filter(Boolean);
|
||||||
|
}
|
||||||
|
|
||||||
export function levenshtein(a, b) {
|
export function levenshtein(a, b) {
|
||||||
const m = a.length, n = b.length;
|
const m = a.length, n = b.length;
|
||||||
if (!m) return n;
|
if (!m) return n;
|
||||||
|
|||||||
Reference in New Issue
Block a user