generated from ztimson/template
139 lines
5.0 KiB
JavaScript
139 lines
5.0 KiB
JavaScript
'use strict';
|
|
|
|
import fs from 'node:fs';
|
|
import {decompressPool} from './decompress.js';
|
|
|
|
const HEADER_SIZE = 80;
|
|
const NS_CONTENT = 'C';
|
|
const NS_METADATA = 'M';
|
|
|
|
async function readAt(fd, pos, length) {
|
|
const buf = Buffer.alloc(length);
|
|
await fd.read(buf, 0, length, pos);
|
|
return buf;
|
|
}
|
|
|
|
async function ptr64(fd, base, index) {
|
|
return Number((await readAt(fd, base + index * 8, 8)).readBigUInt64LE(0));
|
|
}
|
|
|
|
async function readHeader(fd) {
|
|
const b = await readAt(fd, 0, HEADER_SIZE);
|
|
return {
|
|
articleCount: b.readUInt32LE(24),
|
|
clusterCount: b.readUInt32LE(28),
|
|
urlPtrPos: Number(b.readBigUInt64LE(32)),
|
|
clusterPtrPos: Number(b.readBigUInt64LE(48)),
|
|
mimeListPos: Number(b.readBigUInt64LE(56)),
|
|
};
|
|
}
|
|
|
|
async function readMimeTypes(fd, mimeListPos) {
|
|
let pos = mimeListPos, str = '';
|
|
for (;;) {
|
|
str += (await readAt(fd, pos, 1024)).toString('binary');
|
|
const end = str.indexOf('\0\0');
|
|
if (end !== -1) { str = str.slice(0, end + 1); break; }
|
|
pos += 1024;
|
|
}
|
|
return str.split('\0').filter(Boolean);
|
|
}
|
|
|
|
/** Directory entry (article record) at byte `offset`, growing the read window as needed. */
|
|
async function readDirent(fd, offset) {
|
|
for (let size = 512; ; size *= 2) {
|
|
const buf = await readAt(fd, offset, size);
|
|
let o = 0;
|
|
const mimetype = buf.readUInt16LE(o); o += 2;
|
|
o += 1; // extraLen, unused
|
|
const namespace = String.fromCharCode(buf.readUInt8(o)); o += 1;
|
|
o += 4; // revision, unused
|
|
|
|
let redirectIndex = null, cluster = null, blob = null;
|
|
if (mimetype === 0xffff) { redirectIndex = buf.readUInt32LE(o); o += 4; }
|
|
else { cluster = buf.readUInt32LE(o); o += 4; blob = buf.readUInt32LE(o); o += 4; }
|
|
|
|
const urlEnd = buf.indexOf(0, o);
|
|
if (urlEnd === -1) continue;
|
|
const titleEnd = buf.indexOf(0, urlEnd + 1);
|
|
if (titleEnd === -1) continue;
|
|
|
|
const url = buf.toString('utf8', o, urlEnd);
|
|
const title = buf.toString('utf8', urlEnd + 1, titleEnd) || url;
|
|
return {mimetype, namespace, redirectIndex, cluster, blob, url, title};
|
|
}
|
|
}
|
|
|
|
async function findByUrl(fd, header, url, namespace) {
|
|
const key = namespace + url;
|
|
let lo = 0, hi = header.articleCount - 1;
|
|
while (lo <= hi) {
|
|
const mid = (lo + hi) >> 1;
|
|
const dirent = await readDirent(fd, await ptr64(fd, header.urlPtrPos, mid));
|
|
const dirKey = dirent.namespace + dirent.url;
|
|
const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0;
|
|
if (cmp === 0) return dirent;
|
|
if (cmp < 0) hi = mid - 1; else lo = mid + 1;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
async function resolveRedirect(fd, header, dirent) {
|
|
if (dirent.mimetype !== 0xffff) return dirent;
|
|
return readDirent(fd, await ptr64(fd, header.urlPtrPos, dirent.redirectIndex));
|
|
}
|
|
|
|
async function getBlob(fd, header, filePath, clusterNumber, blobNumber) {
|
|
const start = await ptr64(fd, header.clusterPtrPos, clusterNumber);
|
|
const isLast = clusterNumber === header.clusterCount - 1;
|
|
const end = isLast
|
|
? (await fs.promises.stat(filePath)).size
|
|
: await ptr64(fd, header.clusterPtrPos, clusterNumber + 1);
|
|
|
|
const raw = await readAt(fd, start, end - start);
|
|
const compType = raw[0] & 0x0f;
|
|
const extended = (raw[0] & 0x10) !== 0;
|
|
const body = raw.subarray(1);
|
|
|
|
let data;
|
|
if (compType <= 1) data = Buffer.from(body);
|
|
else if (compType === 4 || compType === 5) data = await decompressPool.run({compType, body});
|
|
else throw new Error(`Unsupported cluster compression type: ${compType}`);
|
|
|
|
const readPtr = i => extended ? Number(data.readBigUInt64LE(i * 8)) : data.readUInt32LE(i * 4);
|
|
return data.subarray(readPtr(blobNumber), readPtr(blobNumber + 1));
|
|
}
|
|
|
|
/**
|
|
* One-shot read of a single entry from a .zim archive - no server required.
|
|
* Opt-in convenience for callers who don't want to run kiwix-serve for a single
|
|
* lookup. Re-parses the header/mimetype list on every call; fine for occasional
|
|
* reads, not meant for high-volume access (use KiwixServer for that).
|
|
*/
|
|
export async function readZimEntry(zimPath, url, namespace = NS_CONTENT) {
|
|
const fd = await fs.promises.open(zimPath, 'r');
|
|
try {
|
|
const header = await readHeader(fd);
|
|
const mimeTypes = await readMimeTypes(fd, header.mimeListPos);
|
|
let dirent = await findByUrl(fd, header, url, namespace);
|
|
if (!dirent) return null;
|
|
dirent = await resolveRedirect(fd, header, dirent);
|
|
const data = await getBlob(fd, header, zimPath, dirent.cluster, dirent.blob);
|
|
return {mimetype: mimeTypes[dirent.mimetype] || 'application/octet-stream', data};
|
|
} finally {
|
|
await fd.close();
|
|
}
|
|
}
|
|
|
|
/** Reads 'M' namespace metadata (Title, Creator, Date, etc.) without a server. Minimal by design - no icon/size/counts. */
|
|
export async function readZimMetadata(zimPath) {
|
|
const keys = ['Title', 'Creator', 'Publisher', 'Date', 'Description', 'Language', 'Name', 'Tags'];
|
|
const entries = await Promise.all(keys.map(k => readZimEntry(zimPath, k, NS_METADATA)));
|
|
const [title, creator, publisher, date, description, language, name, tags] = entries.map(e => e?.data.toString('utf8') ?? null);
|
|
return {
|
|
title, updated: date ? new Date(date) : null, summary: description, language, name,
|
|
category: tags ? tags.split(';')[0] || '' : '', tags: tags ? tags.split(';') : [],
|
|
author: creator, publisher,
|
|
};
|
|
}
|