generated from ztimson/template
356 lines
13 KiB
JavaScript
356 lines
13 KiB
JavaScript
'use strict';
|
|
|
|
import fs from 'node:fs';
|
|
import path from 'node:path';
|
|
import {decompressPool} from './decompress.js';
|
|
import {fuzzyMatch, titleFromUrl} from './utils.js';
|
|
|
|
const INDEX_VERSION = 1;
|
|
const HEADER_SIZE = 80;
|
|
const NS_CONTENT = 'C';
|
|
const NS_METADATA = 'M';
|
|
const TITLE_SENTINEL = 0xffffffffffffffffn; // Indicator -> ZIM v6+ archives with no title
|
|
|
|
const DEFAULT_CLUSTER_CACHE_MAX = 32;
|
|
const DEFAULT_CLUSTER_TTL = 60_000;
|
|
|
|
/** Native, dependency-light reader for .zim archives. Supports zstd & LZMA cluster compression. */
|
|
export class ZimReader {
|
|
#fd = null;
|
|
#header = null;
|
|
#mimeTypes = [];
|
|
#hasTitleListing = false;
|
|
#index;
|
|
#clusterCache = new Map(); // clusterNumber -> {data, extended, timer}
|
|
#pending = new Map(); // clusterNumber -> Promise, dedupes concurrent misses
|
|
#clusterCacheMax;
|
|
#clusterTTL;
|
|
|
|
get articleCount() { return this.#header?.articleCount ?? 0; }
|
|
get mediaCount() { return this.#header?.clusterCount ?? 0; }
|
|
|
|
/** @param {{clusterCacheMax?: number, clusterTTL?: number}} [opts] clusterTTL in ms; 0/null disables idle eviction. */
|
|
constructor(path, {clusterCacheMax = DEFAULT_CLUSTER_CACHE_MAX, clusterTTL = DEFAULT_CLUSTER_TTL} = {}) {
|
|
this.path = path;
|
|
this.#clusterCacheMax = clusterCacheMax;
|
|
this.#clusterTTL = clusterTTL;
|
|
}
|
|
|
|
/** Full O(n) scan over the URL pointer list, used when there's no title index. */
|
|
async #allDirents() {
|
|
const dirents = [];
|
|
for (let i = 0; i < this.#header.articleCount; i++) {
|
|
dirents.push(await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, i)));
|
|
}
|
|
return dirents;
|
|
}
|
|
|
|
/** Binary search the URL pointer list for namespace+url. */
|
|
async #findByUrl(url, namespace) {
|
|
const key = namespace + url;
|
|
let lo = 0, hi = this.#header.articleCount - 1;
|
|
while (lo <= hi) {
|
|
const mid = (lo + hi) >> 1;
|
|
const dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, mid));
|
|
const dirKey = dirent.namespace + dirent.url;
|
|
const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0;
|
|
if (cmp === 0) return dirent;
|
|
if (cmp < 0) hi = mid - 1; else lo = mid + 1;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/** Resets a cluster's idle-eviction timer. No-op when TTL disabled. */
|
|
#touch(clusterNumber, entry) {
|
|
if (!this.#clusterTTL) return;
|
|
clearTimeout(entry.timer);
|
|
entry.timer = setTimeout(() => this.#clusterCache.delete(clusterNumber), this.#clusterTTL).unref();
|
|
}
|
|
|
|
/** Fetches + decompresses a cluster exactly once, offloading decompression to the worker pool. */
|
|
async #loadCluster(clusterNumber) {
|
|
const start = await this.#ptr64(this.#header.clusterPtrPos, clusterNumber);
|
|
const isLast = clusterNumber === this.#header.clusterCount - 1;
|
|
const end = isLast
|
|
? (await fs.promises.stat(this.path)).size
|
|
: await this.#ptr64(this.#header.clusterPtrPos, clusterNumber + 1);
|
|
|
|
const raw = await this.#read(start, end - start);
|
|
const compType = raw[0] & 0x0f;
|
|
const extended = (raw[0] & 0x10) !== 0;
|
|
const body = raw.subarray(1);
|
|
|
|
let data;
|
|
if (compType <= 1) data = Buffer.from(body);
|
|
else if (compType === 4 || compType === 5) data = await decompressPool.run({compType, body});
|
|
else throw new Error(`Unsupported cluster compression type: ${compType}`);
|
|
if (!data) throw new Error(`Cluster ${clusterNumber} failed to decompress (compType ${compType})`);
|
|
|
|
return {data, extended};
|
|
}
|
|
|
|
async #getBlob(clusterNumber, blobNumber) {
|
|
let entry = this.#clusterCache.get(clusterNumber);
|
|
if (!entry) {
|
|
let pending = this.#pending.get(clusterNumber);
|
|
if (!pending) {
|
|
pending = this.#loadCluster(clusterNumber);
|
|
this.#pending.set(clusterNumber, pending);
|
|
}
|
|
entry = await pending;
|
|
this.#pending.delete(clusterNumber);
|
|
this.#clusterCache.set(clusterNumber, entry);
|
|
if (this.#clusterCache.size > this.#clusterCacheMax) {
|
|
const oldestKey = this.#clusterCache.keys().next().value;
|
|
clearTimeout(this.#clusterCache.get(oldestKey)?.timer);
|
|
this.#clusterCache.delete(oldestKey);
|
|
}
|
|
}
|
|
this.#touch(clusterNumber, entry);
|
|
|
|
const readPtr = i => entry.extended ? Number(entry.data.readBigUInt64LE(i * 8)) : entry.data.readUInt32LE(i * 4);
|
|
return entry.data.subarray(readPtr(blobNumber), readPtr(blobNumber + 1));
|
|
}
|
|
|
|
async #icon(size = 48) {
|
|
const page = await this.readPage(`Illustration_${size}x${size}@1`, NS_METADATA)
|
|
|| await this.readPage('Favicon', NS_METADATA);
|
|
if (!page) return null;
|
|
return `data:${page.mimetype};base64,${page.data.toString('base64')}`;
|
|
}
|
|
|
|
async #ptr64(base, index) {
|
|
return Number((await this.#read(base + index * 8, 8)).readBigUInt64LE(0));
|
|
}
|
|
|
|
async #read(pos, length) {
|
|
const buf = Buffer.alloc(length);
|
|
await this.#fd.read(buf, 0, length, pos);
|
|
return buf;
|
|
}
|
|
|
|
async #readHeader() {
|
|
const b = await this.#read(0, HEADER_SIZE);
|
|
const titlePtrRaw = b.readBigUInt64LE(40);
|
|
this.#hasTitleListing = titlePtrRaw !== TITLE_SENTINEL;
|
|
this.#header = {
|
|
articleCount: b.readUInt32LE(24),
|
|
clusterCount: b.readUInt32LE(28),
|
|
urlPtrPos: Number(b.readBigUInt64LE(32)),
|
|
titlePtrPos: this.#hasTitleListing ? Number(titlePtrRaw) : null,
|
|
clusterPtrPos: Number(b.readBigUInt64LE(48)),
|
|
mimeListPos: Number(b.readBigUInt64LE(56)),
|
|
mainPage: b.readUInt32LE(64),
|
|
};
|
|
}
|
|
|
|
async #readMimeTypes() {
|
|
let pos = this.#header.mimeListPos, str = '';
|
|
for (;;) {
|
|
str += (await this.#read(pos, 1024)).toString('binary');
|
|
const end = str.indexOf('\0\0');
|
|
if (end !== -1) { str = str.slice(0, end + 1); break; }
|
|
pos += 1024;
|
|
}
|
|
this.#mimeTypes = str.split('\0').filter(Boolean);
|
|
}
|
|
|
|
/** Directory entry (article record) at byte `offset`, growing the read window as needed. */
|
|
async #readDirent(offset) {
|
|
for (let size = 512; ; size *= 2) {
|
|
const buf = await this.#read(offset, size);
|
|
let o = 0;
|
|
const mimetype = buf.readUInt16LE(o); o += 2;
|
|
o += 1; // extraLen, unused
|
|
const namespace = String.fromCharCode(buf.readUInt8(o)); o += 1;
|
|
o += 4; // revision, unused
|
|
|
|
let redirectIndex = null, cluster = null, blob = null;
|
|
if (mimetype === 0xffff) { redirectIndex = buf.readUInt32LE(o); o += 4; }
|
|
else { cluster = buf.readUInt32LE(o); o += 4; blob = buf.readUInt32LE(o); o += 4; }
|
|
|
|
const urlEnd = buf.indexOf(0, o);
|
|
if (urlEnd === -1) continue;
|
|
const titleEnd = buf.indexOf(0, urlEnd + 1);
|
|
if (titleEnd === -1) continue;
|
|
|
|
const url = buf.toString('utf8', o, urlEnd);
|
|
const title = buf.toString('utf8', urlEnd + 1, titleEnd) || url;
|
|
return {mimetype, namespace, redirectIndex, cluster, blob, url, title};
|
|
}
|
|
}
|
|
|
|
async #resolveRedirect(dirent) {
|
|
if (dirent.mimetype !== 0xffff) return dirent;
|
|
return this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, dirent.redirectIndex));
|
|
}
|
|
|
|
/** Narrows to dirents near the term's alphabetical position in the title index. */
|
|
async #titleIndexCandidates(term) {
|
|
const q = term.toLowerCase();
|
|
let lo = 0, hi = this.#header.articleCount - 1;
|
|
while (lo < hi) {
|
|
const mid = (lo + hi) >> 1;
|
|
const urlIdx = (await this.#read(this.#header.titlePtrPos + mid * 4, 4)).readUInt32LE(0);
|
|
const dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx));
|
|
if (dirent.title.toLowerCase() < q) lo = mid + 1; else hi = mid;
|
|
}
|
|
// Widen around the prefix match since fuzzy scoring isn't purely alphabetical.
|
|
const start = Math.max(0, lo - 50), end = Math.min(this.#header.articleCount, lo + 200);
|
|
const dirents = [];
|
|
for (let i = start; i < end; i++) {
|
|
const urlIdx = (await this.#read(this.#header.titlePtrPos + i * 4, 4)).readUInt32LE(0);
|
|
dirents.push(await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx)));
|
|
}
|
|
return dirents;
|
|
}
|
|
|
|
/** Path of the cached word index, kept in an `index/` subdirectory next to the archive. */
|
|
#indexPath() {
|
|
return path.join(path.dirname(this.path), 'index', `${path.basename(this.path)}.idx.json`);
|
|
}
|
|
|
|
/** Builds (or loads a cached) word -> entry index, avoiding a re-scan on every search. */
|
|
async #wordIndex() {
|
|
if (this.#index) return this.#index;
|
|
const cachePath = this.#indexPath();
|
|
const stat = await fs.promises.stat(this.path);
|
|
|
|
if (fs.existsSync(cachePath)) {
|
|
try {
|
|
const cached = JSON.parse(await fs.promises.readFile(cachePath, 'utf8'));
|
|
// Rebuild if stale: index predates the archive's current mtime (e.g. a re-download).
|
|
if (cached.version === INDEX_VERSION && cached.mtimeMs >= stat.mtimeMs) {
|
|
return (this.#index = cached);
|
|
}
|
|
} catch {}
|
|
}
|
|
|
|
// Expensive full scan — only happens once per archive (or after it changes).
|
|
const dirents = await this.#allDirents();
|
|
const entries = [];
|
|
const words = new Map(); // word -> [entryIndex, ...]
|
|
for (const d of dirents) {
|
|
if (d.namespace !== NS_CONTENT) continue;
|
|
const idx = entries.length;
|
|
entries.push({url: d.url, title: d.title, mimetype: d.mimetype});
|
|
const text = `${d.title} ${titleFromUrl(d.url)}`.toLowerCase();
|
|
for (const w of text.split(/\W+/).filter(Boolean)) {
|
|
if (!words.has(w)) words.set(w, []);
|
|
words.get(w).push(idx);
|
|
}
|
|
}
|
|
|
|
const index = {version: INDEX_VERSION, size: stat.size, mtimeMs: stat.mtimeMs, entries, words: Object.fromEntries(words)};
|
|
await fs.promises.mkdir(path.dirname(cachePath), {recursive: true});
|
|
await fs.promises.writeFile(cachePath, JSON.stringify(index));
|
|
return (this.#index = index);
|
|
}
|
|
|
|
async close() {
|
|
if (this.#fd) await this.#fd.close();
|
|
this.#fd = null;
|
|
for (const entry of this.#clusterCache.values()) clearTimeout(entry.timer);
|
|
this.#clusterCache.clear();
|
|
this.#pending.clear();
|
|
}
|
|
|
|
/** Deletes the zim archive and its cached index (if any). Safe to call on unopened readers. */
|
|
async delete() {
|
|
await this.close();
|
|
this.#index = null;
|
|
await fs.promises.rm(this.path, {force: true});
|
|
await fs.promises.rm(this.#indexPath(), {force: true});
|
|
}
|
|
|
|
/** Reads an 'M' namespace metadata value (e.g. Name, Date, Title). Returns null if missing. */
|
|
async metadata(key) {
|
|
const get = async key => {
|
|
const page = await this.readPage(key, NS_METADATA);
|
|
return page ? page.data.toString('utf8') : null;
|
|
};
|
|
if(key) return get(key);
|
|
|
|
const [title, creator, publisher, date, description, language, name, tags] = await Promise.all(
|
|
['Title', 'Creator', 'Publisher', 'Date', 'Description', 'Language', 'Name', 'Tags'].map(get)
|
|
);
|
|
return {
|
|
title,
|
|
updated: date ? new Date(date) : null,
|
|
summary: description,
|
|
language,
|
|
name,
|
|
category: tags ? tags.split(';')[0] || '' : '',
|
|
tags: tags ? tags.split(';') : [],
|
|
author: creator,
|
|
publisher,
|
|
articleCount: this.articleCount,
|
|
mediaCount: this.mediaCount,
|
|
sizeMb: +((await fs.promises.stat(this.path)).size / 1024 / 1024).toFixed(1),
|
|
icon: await this.#icon(),
|
|
};
|
|
}
|
|
|
|
/** Opens the archive and parses its header + mimetype list. */
|
|
async open() {
|
|
this.#fd = await fs.promises.open(this.path, 'r');
|
|
await this.#readHeader();
|
|
await this.#readMimeTypes();
|
|
return this;
|
|
}
|
|
|
|
/** Read a page's content by URL. Returns `{mimetype, data}` or `null` if not found. */
|
|
async readPage(url, namespace = NS_CONTENT) {
|
|
let dirent = await this.#findByUrl(url, namespace);
|
|
if (!dirent) return null;
|
|
dirent = await this.#resolveRedirect(dirent);
|
|
const data = await this.#getBlob(dirent.cluster, dirent.blob);
|
|
return {mimetype: this.#mimeTypes[dirent.mimetype] || 'application/octet-stream', data};
|
|
}
|
|
|
|
/** Reads the archive's designated main/landing page, if one is set. */
|
|
async mainPage() {
|
|
if (this.#header.mainPage === 0xffffffff) return null;
|
|
let dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, this.#header.mainPage));
|
|
dirent = await this.#resolveRedirect(dirent);
|
|
const data = await this.#getBlob(dirent.cluster, dirent.blob);
|
|
return {mimetype: this.#mimeTypes[dirent.mimetype] || 'application/octet-stream', data, url: dirent.url, namespace: dirent.namespace};
|
|
}
|
|
|
|
/**
|
|
* Fuzzy-ranked title search. Accepts comma-separated `terms` the same way the
|
|
* catalog search does. Uses the sorted title index when present (binary search
|
|
* narrows the candidate window); falls back to a full linear scan otherwise
|
|
* (common on ZIM v6+/zimit-generated archives with no title index).
|
|
*/
|
|
async search(terms, {limit = 20, htmlOnly = true} = {}) {
|
|
const termList = String(terms).split(',').map(t => t.trim()).filter(Boolean);
|
|
if (!termList.length) return [];
|
|
|
|
let candidates;
|
|
if (this.#hasTitleListing && this.#header.articleCount > 5000) {
|
|
candidates = await this.#titleIndexCandidates(termList[0]);
|
|
} else {
|
|
const {entries, words} = await this.#wordIndex();
|
|
// Pull candidates from postings of any word that starts with (or contains) the search term.
|
|
const q = termList[0].toLowerCase();
|
|
const idxSet = new Set();
|
|
for (const [word, postings] of Object.entries(words)) {
|
|
if (word.includes(q) || q.includes(word)) postings.forEach(i => idxSet.add(i));
|
|
}
|
|
candidates = [...idxSet].map(i => ({...entries[i], namespace: NS_CONTENT}));
|
|
}
|
|
|
|
const scored = [];
|
|
for (const dirent of candidates) {
|
|
if (htmlOnly && !(this.#mimeTypes[dirent.mimetype] || '').startsWith('text/html')) continue;
|
|
const urlTitle = titleFromUrl(dirent.url);
|
|
const titleScore = fuzzyMatch(dirent.title, ...termList).max;
|
|
const urlScore = fuzzyMatch(urlTitle, ...termList).max;
|
|
scored.push({url: dirent.url, title: dirent.title.length > urlTitle.length ? dirent.title : urlTitle, namespace: NS_CONTENT, score: Math.max(titleScore, urlScore)});
|
|
}
|
|
const {summary, mediaCount, articleCount, sizeMb, ...meta} = await this.metadata();
|
|
return scored.filter(a => a.score > 0).toSorted((a, b) => b.score - a.score).slice(0, limit).map(a => ({...meta, ...a}));
|
|
}
|
|
}
|