diff --git a/package.json b/package.json index bd0fabc..f4bcb14 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@ztimson/zim-utils", - "version": "0.2.4", + "version": "0.2.5", "description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js", "author": "Zak Timson", "license": "MIT", diff --git a/src/reader.js b/src/reader.js index 012da08..65c24fd 100644 --- a/src/reader.js +++ b/src/reader.js @@ -1,6 +1,7 @@ 'use strict'; import fs from 'node:fs'; +import {createHash} from 'node:crypto'; import {decompressPool} from './decompress.js'; import {fuzzyMatch, titleFromUrl, kiwixSearch} from './utils.js'; @@ -12,6 +13,12 @@ const TITLE_SENTINEL = 0xffffffffffffffffn; // Indicator -> ZIM v6+ archives wit const DEFAULT_CLUSTER_CACHE_MAX = 32; const DEFAULT_CLUSTER_TTL = 60_000; +// --- Fallback search index (only used for archives with no native title listing) --- +const INDEX_SUFFIX = '.searchidx.bin'; +const INDEX_MAGIC = 'ZXI1'; +const INDEX_HEADER_SIZE = 24; // magic(4) + staleness key(16) + recordCount(4) +const INDEX_RECORD_SIZE = 20; // keyOff(4) keyLen(2) urlOff(4) urlLen(2) titleOff(4) titleLen(2) mimetype(2) + /** Native, dependency-light reader for .zim archives. Supports zstd & LZMA cluster compression. */ export class ZimReader { #fd = null; @@ -23,6 +30,12 @@ export class ZimReader { #clusterCacheMax; #clusterTTL; + #indexPath; + #indexReady = null; // Promise, awaited by search() before using the fallback index + #indexFd = null; // open fd for the fallback index, once loaded/built + #indexRecordCount = 0; + #indexTableStart = 0; + get articleCount() { return this.#header?.articleCount ?? 0; } get mediaCount() { return this.#header?.clusterCount ?? 0; } @@ -31,9 +44,10 @@ export class ZimReader { this.path = path; this.#clusterCacheMax = clusterCacheMax; this.#clusterTTL = clusterTTL; + this.#indexPath = `${path}${INDEX_SUFFIX}`; } - /** Binary search the URL pointer list for namespace+url. */ + /** Binary search the URL pointer list for namespace+url. For exact-key lookups (readPage, metadata, icons). */ async #findByUrl(url, namespace) { const key = namespace + url; let lo = 0, hi = this.#header.articleCount - 1; @@ -48,18 +62,178 @@ export class ZimReader { return null; } - /** Binary search the title index for an exact title match. */ - async #findByTitle(title) { - if (!this.#hasTitleListing) return null; - const q = title.toLowerCase(); + /** Binary search the ZIM's own title pointer list. O(log n), zero extra storage - the happy path. */ + async #findByTitleBuiltin(title) { let lo = 0, hi = this.#header.articleCount - 1; while (lo <= hi) { const mid = (lo + hi) >> 1; const urlIdx = (await this.#read(this.#header.titlePtrPos + mid * 4, 4)).readUInt32LE(0); - const dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx)); - const t = dirent.title.toLowerCase(); - if (t === q) return dirent; - if (t < q) lo = mid + 1; else hi = mid - 1; + const d = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx)); + if (d.title === title) return d; + if (d.title < title) lo = mid + 1; else hi = mid - 1; + } + return null; + } + + /** + * Resolves a kiwix-search result (a title, not necessarily a url) to + * {url, title, mimetype}. kiwix-search's fulltext index returns titles, and + * a title isn't guaranteed to equal its url (unicode normalization, + * disambiguation suffixes, punctuation stripping), so: + * 1. URL binary search - matches when title happens to equal url (free to check) + * 2. Title pointer list - when the archive ships one (Wikipedia etc. do) + * 3. Persisted fallback index / linear scan - only for archives without (2) + */ + async #resolveSearchEntry(name) { + let dirent = await this.#findByUrl(name, NS_CONTENT); + if (dirent) return dirent; + + if (this.#hasTitleListing) return this.#findByTitleBuiltin(name); + + if (this.#indexReady) await this.#indexReady; + const hit = await this.#lookupFallbackIndex(name); + if (hit) return hit; + + // Index unavailable (build failed - unwritable disk, etc.) or genuinely no match. + for (let i = 0; i < this.#header.articleCount; i++) { + const d = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, i)); + if (d.namespace === NS_CONTENT && (d.title === name || titleFromUrl(d.url) === name)) return d; + } + return null; + } + + /** + * A cheap, stable fingerprint for "is the cached index still valid for this file". + * ZIM archives end with a 16-byte MD5 of their own contents (the same trailer + * `zimcheck` validates against), so this is a single 16-byte read regardless of + * archive size - no need to hash a multi-hundred-GB file. It's also + * content-based rather than path/mtime-based, so moving or redownloading an + * identical archive doesn't invalidate the cache. Falls back to a tiny + * size+mtime hash only if the file is too short to have a real trailer. + */ + async #stalenessKey() { + const {size} = await fs.promises.stat(this.path); + if (size >= 16) return this.#read(size - 16, 16); + const stat = await fs.promises.stat(this.path); + return createHash('md5').update(`${stat.size}:${stat.mtimeMs}`).digest(); + } + + /** Walks every content dirent once, resolving redirects, keyed by both its title and its url-derived title. */ + async #collectFallbackEntries() { + const map = new Map(); // key -> {url, title, mimetype} + for (let i = 0; i < this.#header.articleCount; i++) { + let dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, i)); + if (dirent.namespace !== NS_CONTENT) continue; + dirent = await this.#resolveRedirect(dirent); + if (dirent.namespace !== NS_CONTENT) continue; // redirected outside content namespace + + const entry = {url: dirent.url, title: dirent.title, mimetype: dirent.mimetype}; + if (!map.has(dirent.title)) map.set(dirent.title, entry); + const urlTitle = titleFromUrl(dirent.url); + if (urlTitle !== dirent.title && !map.has(urlTitle)) map.set(urlTitle, entry); + } + return map; + } + + /** Serializes the fallback index to disk: header, sorted fixed-width records, then a string table. */ + async #buildFallbackIndex(stalenessKey) { + const map = await this.#collectFallbackEntries(); + const keys = [...map.keys()].sort(); + + const records = Buffer.alloc(keys.length * INDEX_RECORD_SIZE); + const strings = []; + let tableOffset = 0; + keys.forEach((key, i) => { + const {url, title, mimetype} = map.get(key); + const keyBuf = Buffer.from(key, 'utf8'); + const urlBuf = Buffer.from(url, 'utf8'); + const titleBuf = Buffer.from(title, 'utf8'); + const base = i * INDEX_RECORD_SIZE; + + records.writeUInt32LE(tableOffset, base); records.writeUInt16LE(keyBuf.length, base + 4); + tableOffset += keyBuf.length; + records.writeUInt32LE(tableOffset, base + 6); records.writeUInt16LE(urlBuf.length, base + 10); + tableOffset += urlBuf.length; + records.writeUInt32LE(tableOffset, base + 12); records.writeUInt16LE(titleBuf.length, base + 16); + tableOffset += titleBuf.length; + records.writeUInt16LE(mimetype, base + 18); + + strings.push(keyBuf, urlBuf, titleBuf); + }); + + const header = Buffer.alloc(INDEX_HEADER_SIZE); + header.write(INDEX_MAGIC, 0, 'ascii'); + stalenessKey.copy(header, 4); + header.writeUInt32LE(keys.length, 20); + + const tmpPath = `${this.#indexPath}.tmp-${process.pid}`; + await fs.promises.writeFile(tmpPath, Buffer.concat([header, records, ...strings])); + await fs.promises.rename(tmpPath, this.#indexPath); + } + + /** Opens the fallback index file and caches its header fields for querying. */ + async #openFallbackIndex() { + const fd = await fs.promises.open(this.#indexPath, 'r'); + const header = Buffer.alloc(INDEX_HEADER_SIZE); + await fd.read(header, 0, INDEX_HEADER_SIZE, 0); + this.#indexFd = fd; + this.#indexRecordCount = header.readUInt32LE(20); + this.#indexTableStart = INDEX_HEADER_SIZE + this.#indexRecordCount * INDEX_RECORD_SIZE; + } + + /** + * Validates the on-disk fallback index against the archive's current staleness + * key and (re)builds it if missing/stale. Failures are swallowed - callers see + * #indexFd stay null and fall through to the linear-scan last resort. + */ + async #loadOrBuildFallbackIndex() { + const stalenessKey = await this.#stalenessKey(); + + try { + const fd = await fs.promises.open(this.#indexPath, 'r'); + const header = Buffer.alloc(INDEX_HEADER_SIZE); + await fd.read(header, 0, INDEX_HEADER_SIZE, 0); + const fresh = header.toString('ascii', 0, 4) === INDEX_MAGIC && header.subarray(4, 20).equals(stalenessKey); + if (fresh) { + this.#indexFd = fd; + this.#indexRecordCount = header.readUInt32LE(20); + this.#indexTableStart = INDEX_HEADER_SIZE + this.#indexRecordCount * INDEX_RECORD_SIZE; + return; + } + await fd.close(); + } catch { /* missing, corrupt, or unreadable -> rebuild below */ } + + await this.#buildFallbackIndex(stalenessKey); + await this.#openFallbackIndex(); + } + + async #indexRead(pos, length) { + const buf = Buffer.alloc(length); + await this.#indexFd.read(buf, 0, length, pos); + return buf; + } + + async #indexString(offset, length) { + if (!length) return ''; + return (await this.#indexRead(this.#indexTableStart + offset, length)).toString('utf8'); + } + + /** Binary search over the on-disk record array. Never loads the full index into memory. */ + async #lookupFallbackIndex(name) { + if (!this.#indexFd) return null; + let lo = 0, hi = this.#indexRecordCount - 1; + while (lo <= hi) { + const mid = (lo + hi) >> 1; + const rec = await this.#indexRead(INDEX_HEADER_SIZE + mid * INDEX_RECORD_SIZE, INDEX_RECORD_SIZE); + const keyOff = rec.readUInt32LE(0), keyLen = rec.readUInt16LE(4); + const key = await this.#indexString(keyOff, keyLen); + if (key === name) { + const urlOff = rec.readUInt32LE(6), urlLen = rec.readUInt16LE(10); + const titleOff = rec.readUInt32LE(12), titleLen = rec.readUInt16LE(16); + const [url, title] = await Promise.all([this.#indexString(urlOff, urlLen), this.#indexString(titleOff, titleLen)]); + return {url, title, mimetype: rec.readUInt16LE(18)}; + } + if (key < name) lo = mid + 1; else hi = mid - 1; } return null; } @@ -192,15 +366,18 @@ export class ZimReader { async close() { if (this.#fd) await this.#fd.close(); this.#fd = null; + if (this.#indexFd) await this.#indexFd.close(); + this.#indexFd = null; for (const entry of this.#clusterCache.values()) clearTimeout(entry.timer); this.#clusterCache.clear(); this.#pending.clear(); } - /** Deletes the zim archive. Safe to call on unopened readers. */ + /** Deletes the zim archive and its cached fallback index (if any). Safe to call on unopened readers. */ async delete() { await this.close(); await fs.promises.rm(this.path, {force: true}); + await fs.promises.rm(this.#indexPath, {force: true}); } /** Reads an 'M' namespace metadata value (e.g. Name, Date, Title). Returns null if missing. */ @@ -242,6 +419,14 @@ export class ZimReader { this.#fd = null; throw e; } + + // The fallback index is only needed when the archive has no native title + // pointer list - #findByTitleBuiltin already covers that case in O(log n) + // with zero extra storage. Well-maintained archives (Wikipedia etc.) + // ship a title listing, so this path is expected to be rare in practice. + if (!this.#hasTitleListing) { + this.#indexReady = this.#loadOrBuildFallbackIndex().catch(() => {}); + } return this; } @@ -272,8 +457,8 @@ export class ZimReader { const termList = String(terms).split(',').map(t => t.trim()).filter(Boolean); if (!termList.length) return []; - const titles = await kiwixSearch(this.path, termList.join(' ')); - const candidates = (await Promise.all(titles.map(t => this.#findByTitle(t)))).filter(Boolean); + const names = await kiwixSearch(this.path, termList.join(' ')); + const candidates = (await Promise.all(names.map(n => this.#resolveSearchEntry(n)))).filter(Boolean); const scored = []; for (const dirent of candidates) {