Create title index for zims missing them
All checks were successful
Publish Library / Build NPM Project (push) Successful in 17s
Publish Library / Tag Version (push) Successful in 10s

This commit is contained in:
2026-08-24 13:08:17 -04:00
parent bfb3f0efd3
commit 7d376c90b4
2 changed files with 198 additions and 13 deletions

View File

@@ -1,6 +1,7 @@
'use strict';
import fs from 'node:fs';
import {createHash} from 'node:crypto';
import {decompressPool} from './decompress.js';
import {fuzzyMatch, titleFromUrl, kiwixSearch} from './utils.js';
@@ -12,6 +13,12 @@ const TITLE_SENTINEL = 0xffffffffffffffffn; // Indicator -> ZIM v6+ archives wit
const DEFAULT_CLUSTER_CACHE_MAX = 32;
const DEFAULT_CLUSTER_TTL = 60_000;
// --- Fallback search index (only used for archives with no native title listing) ---
const INDEX_SUFFIX = '.searchidx.bin';
const INDEX_MAGIC = 'ZXI1';
const INDEX_HEADER_SIZE = 24; // magic(4) + staleness key(16) + recordCount(4)
const INDEX_RECORD_SIZE = 20; // keyOff(4) keyLen(2) urlOff(4) urlLen(2) titleOff(4) titleLen(2) mimetype(2)
/** Native, dependency-light reader for .zim archives. Supports zstd & LZMA cluster compression. */
export class ZimReader {
#fd = null;
@@ -23,6 +30,12 @@ export class ZimReader {
#clusterCacheMax;
#clusterTTL;
#indexPath;
#indexReady = null; // Promise, awaited by search() before using the fallback index
#indexFd = null; // open fd for the fallback index, once loaded/built
#indexRecordCount = 0;
#indexTableStart = 0;
get articleCount() { return this.#header?.articleCount ?? 0; }
get mediaCount() { return this.#header?.clusterCount ?? 0; }
@@ -31,9 +44,10 @@ export class ZimReader {
this.path = path;
this.#clusterCacheMax = clusterCacheMax;
this.#clusterTTL = clusterTTL;
this.#indexPath = `${path}${INDEX_SUFFIX}`;
}
/** Binary search the URL pointer list for namespace+url. */
/** Binary search the URL pointer list for namespace+url. For exact-key lookups (readPage, metadata, icons). */
async #findByUrl(url, namespace) {
const key = namespace + url;
let lo = 0, hi = this.#header.articleCount - 1;
@@ -48,18 +62,178 @@ export class ZimReader {
return null;
}
/** Binary search the title index for an exact title match. */
async #findByTitle(title) {
if (!this.#hasTitleListing) return null;
const q = title.toLowerCase();
/** Binary search the ZIM's own title pointer list. O(log n), zero extra storage - the happy path. */
async #findByTitleBuiltin(title) {
let lo = 0, hi = this.#header.articleCount - 1;
while (lo <= hi) {
const mid = (lo + hi) >> 1;
const urlIdx = (await this.#read(this.#header.titlePtrPos + mid * 4, 4)).readUInt32LE(0);
const dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx));
const t = dirent.title.toLowerCase();
if (t === q) return dirent;
if (t < q) lo = mid + 1; else hi = mid - 1;
const d = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx));
if (d.title === title) return d;
if (d.title < title) lo = mid + 1; else hi = mid - 1;
}
return null;
}
/**
* Resolves a kiwix-search result (a title, not necessarily a url) to
* {url, title, mimetype}. kiwix-search's fulltext index returns titles, and
* a title isn't guaranteed to equal its url (unicode normalization,
* disambiguation suffixes, punctuation stripping), so:
* 1. URL binary search - matches when title happens to equal url (free to check)
* 2. Title pointer list - when the archive ships one (Wikipedia etc. do)
* 3. Persisted fallback index / linear scan - only for archives without (2)
*/
async #resolveSearchEntry(name) {
let dirent = await this.#findByUrl(name, NS_CONTENT);
if (dirent) return dirent;
if (this.#hasTitleListing) return this.#findByTitleBuiltin(name);
if (this.#indexReady) await this.#indexReady;
const hit = await this.#lookupFallbackIndex(name);
if (hit) return hit;
// Index unavailable (build failed - unwritable disk, etc.) or genuinely no match.
for (let i = 0; i < this.#header.articleCount; i++) {
const d = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, i));
if (d.namespace === NS_CONTENT && (d.title === name || titleFromUrl(d.url) === name)) return d;
}
return null;
}
/**
* A cheap, stable fingerprint for "is the cached index still valid for this file".
* ZIM archives end with a 16-byte MD5 of their own contents (the same trailer
* `zimcheck` validates against), so this is a single 16-byte read regardless of
* archive size - no need to hash a multi-hundred-GB file. It's also
* content-based rather than path/mtime-based, so moving or redownloading an
* identical archive doesn't invalidate the cache. Falls back to a tiny
* size+mtime hash only if the file is too short to have a real trailer.
*/
async #stalenessKey() {
const {size} = await fs.promises.stat(this.path);
if (size >= 16) return this.#read(size - 16, 16);
const stat = await fs.promises.stat(this.path);
return createHash('md5').update(`${stat.size}:${stat.mtimeMs}`).digest();
}
/** Walks every content dirent once, resolving redirects, keyed by both its title and its url-derived title. */
async #collectFallbackEntries() {
const map = new Map(); // key -> {url, title, mimetype}
for (let i = 0; i < this.#header.articleCount; i++) {
let dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, i));
if (dirent.namespace !== NS_CONTENT) continue;
dirent = await this.#resolveRedirect(dirent);
if (dirent.namespace !== NS_CONTENT) continue; // redirected outside content namespace
const entry = {url: dirent.url, title: dirent.title, mimetype: dirent.mimetype};
if (!map.has(dirent.title)) map.set(dirent.title, entry);
const urlTitle = titleFromUrl(dirent.url);
if (urlTitle !== dirent.title && !map.has(urlTitle)) map.set(urlTitle, entry);
}
return map;
}
/** Serializes the fallback index to disk: header, sorted fixed-width records, then a string table. */
async #buildFallbackIndex(stalenessKey) {
const map = await this.#collectFallbackEntries();
const keys = [...map.keys()].sort();
const records = Buffer.alloc(keys.length * INDEX_RECORD_SIZE);
const strings = [];
let tableOffset = 0;
keys.forEach((key, i) => {
const {url, title, mimetype} = map.get(key);
const keyBuf = Buffer.from(key, 'utf8');
const urlBuf = Buffer.from(url, 'utf8');
const titleBuf = Buffer.from(title, 'utf8');
const base = i * INDEX_RECORD_SIZE;
records.writeUInt32LE(tableOffset, base); records.writeUInt16LE(keyBuf.length, base + 4);
tableOffset += keyBuf.length;
records.writeUInt32LE(tableOffset, base + 6); records.writeUInt16LE(urlBuf.length, base + 10);
tableOffset += urlBuf.length;
records.writeUInt32LE(tableOffset, base + 12); records.writeUInt16LE(titleBuf.length, base + 16);
tableOffset += titleBuf.length;
records.writeUInt16LE(mimetype, base + 18);
strings.push(keyBuf, urlBuf, titleBuf);
});
const header = Buffer.alloc(INDEX_HEADER_SIZE);
header.write(INDEX_MAGIC, 0, 'ascii');
stalenessKey.copy(header, 4);
header.writeUInt32LE(keys.length, 20);
const tmpPath = `${this.#indexPath}.tmp-${process.pid}`;
await fs.promises.writeFile(tmpPath, Buffer.concat([header, records, ...strings]));
await fs.promises.rename(tmpPath, this.#indexPath);
}
/** Opens the fallback index file and caches its header fields for querying. */
async #openFallbackIndex() {
const fd = await fs.promises.open(this.#indexPath, 'r');
const header = Buffer.alloc(INDEX_HEADER_SIZE);
await fd.read(header, 0, INDEX_HEADER_SIZE, 0);
this.#indexFd = fd;
this.#indexRecordCount = header.readUInt32LE(20);
this.#indexTableStart = INDEX_HEADER_SIZE + this.#indexRecordCount * INDEX_RECORD_SIZE;
}
/**
* Validates the on-disk fallback index against the archive's current staleness
* key and (re)builds it if missing/stale. Failures are swallowed - callers see
* #indexFd stay null and fall through to the linear-scan last resort.
*/
async #loadOrBuildFallbackIndex() {
const stalenessKey = await this.#stalenessKey();
try {
const fd = await fs.promises.open(this.#indexPath, 'r');
const header = Buffer.alloc(INDEX_HEADER_SIZE);
await fd.read(header, 0, INDEX_HEADER_SIZE, 0);
const fresh = header.toString('ascii', 0, 4) === INDEX_MAGIC && header.subarray(4, 20).equals(stalenessKey);
if (fresh) {
this.#indexFd = fd;
this.#indexRecordCount = header.readUInt32LE(20);
this.#indexTableStart = INDEX_HEADER_SIZE + this.#indexRecordCount * INDEX_RECORD_SIZE;
return;
}
await fd.close();
} catch { /* missing, corrupt, or unreadable -> rebuild below */ }
await this.#buildFallbackIndex(stalenessKey);
await this.#openFallbackIndex();
}
async #indexRead(pos, length) {
const buf = Buffer.alloc(length);
await this.#indexFd.read(buf, 0, length, pos);
return buf;
}
async #indexString(offset, length) {
if (!length) return '';
return (await this.#indexRead(this.#indexTableStart + offset, length)).toString('utf8');
}
/** Binary search over the on-disk record array. Never loads the full index into memory. */
async #lookupFallbackIndex(name) {
if (!this.#indexFd) return null;
let lo = 0, hi = this.#indexRecordCount - 1;
while (lo <= hi) {
const mid = (lo + hi) >> 1;
const rec = await this.#indexRead(INDEX_HEADER_SIZE + mid * INDEX_RECORD_SIZE, INDEX_RECORD_SIZE);
const keyOff = rec.readUInt32LE(0), keyLen = rec.readUInt16LE(4);
const key = await this.#indexString(keyOff, keyLen);
if (key === name) {
const urlOff = rec.readUInt32LE(6), urlLen = rec.readUInt16LE(10);
const titleOff = rec.readUInt32LE(12), titleLen = rec.readUInt16LE(16);
const [url, title] = await Promise.all([this.#indexString(urlOff, urlLen), this.#indexString(titleOff, titleLen)]);
return {url, title, mimetype: rec.readUInt16LE(18)};
}
if (key < name) lo = mid + 1; else hi = mid - 1;
}
return null;
}
@@ -192,15 +366,18 @@ export class ZimReader {
async close() {
if (this.#fd) await this.#fd.close();
this.#fd = null;
if (this.#indexFd) await this.#indexFd.close();
this.#indexFd = null;
for (const entry of this.#clusterCache.values()) clearTimeout(entry.timer);
this.#clusterCache.clear();
this.#pending.clear();
}
/** Deletes the zim archive. Safe to call on unopened readers. */
/** Deletes the zim archive and its cached fallback index (if any). Safe to call on unopened readers. */
async delete() {
await this.close();
await fs.promises.rm(this.path, {force: true});
await fs.promises.rm(this.#indexPath, {force: true});
}
/** Reads an 'M' namespace metadata value (e.g. Name, Date, Title). Returns null if missing. */
@@ -242,6 +419,14 @@ export class ZimReader {
this.#fd = null;
throw e;
}
// The fallback index is only needed when the archive has no native title
// pointer list - #findByTitleBuiltin already covers that case in O(log n)
// with zero extra storage. Well-maintained archives (Wikipedia etc.)
// ship a title listing, so this path is expected to be rare in practice.
if (!this.#hasTitleListing) {
this.#indexReady = this.#loadOrBuildFallbackIndex().catch(() => {});
}
return this;
}
@@ -272,8 +457,8 @@ export class ZimReader {
const termList = String(terms).split(',').map(t => t.trim()).filter(Boolean);
if (!termList.length) return [];
const titles = await kiwixSearch(this.path, termList.join(' '));
const candidates = (await Promise.all(titles.map(t => this.#findByTitle(t)))).filter(Boolean);
const names = await kiwixSearch(this.path, termList.join(' '));
const candidates = (await Promise.all(names.map(n => this.#resolveSearchEntry(n)))).filter(Boolean);
const scored = [];
for (const dirent of candidates) {