generated from ztimson/template
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
943280f326 | ||
|
|
dc24f48a36 | ||
|
|
812e72a8cd | ||
|
|
a45d7bad8c | ||
|
|
0a7e87f6b7 | ||
|
|
e12c3e3cc6 |
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@ztimson/zim-utils",
|
||||
"version": "0.4.0",
|
||||
"version": "0.4.5",
|
||||
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
|
||||
"author": "Zak Timson",
|
||||
"license": "MIT",
|
||||
|
||||
+102
-22
@@ -2,19 +2,29 @@ import {fuzzyMatch} from './utils.js';
|
||||
import {decodeHtml} from '@ztimson/utils';
|
||||
|
||||
export const CATALOG_URL = 'https://library.kiwix.org';
|
||||
export const VIEWER_URL = 'https://browse.library.kiwix.org';
|
||||
|
||||
const PAGE_SIZE = 100;
|
||||
|
||||
export function parseEntries(xml, catalog = CATALOG_URL) {
|
||||
const blocks = xml.match(/<entry>[\s\S]*?<\/entry>/g) || [];
|
||||
|
||||
return blocks.map(b => {
|
||||
const grab = re => (b.match(re) || [])[1] || '';
|
||||
const linkMatch = b.match(/<link[^>]*type=["']application\/x-zim[^"']*["'][^>]*href=["']([^"']+)["']/);
|
||||
const iconMatch = b.match(/<link[^>]*rel=["']http:\/\/opds-spec\.org\/image\/thumbnail["'][^>]*href=["']([^"']+)["']/)
|
||||
|| b.match(/<link[^>]*type=["']image\/[^"']*["'][^>]*href=["']([^"']+)["']/);
|
||||
|
||||
let tags = grab(/<tags>([^<]*)<\/tags>/);
|
||||
if(tags) tags = tags.split(';');
|
||||
if (tags) tags = tags.split(';');
|
||||
|
||||
const name = grab(/<name>([^<]*)<\/name>/);
|
||||
const href = linkMatch ? linkMatch[1] : null;
|
||||
const filename = href
|
||||
? decodeURIComponent(new URL(href).pathname.split('/').pop()).replace(/\.meta4$/i, '')
|
||||
: name;
|
||||
const viewerName = filename.replace(/\.zim$/i, '');
|
||||
|
||||
return {
|
||||
id: grab(/<id>([^<]*)<\/id>/),
|
||||
title: decodeHtml(grab(/<title>([^<]*)<\/title>/)),
|
||||
@@ -22,25 +32,52 @@ export function parseEntries(xml, catalog = CATALOG_URL) {
|
||||
summary: decodeHtml(grab(/<summary>([^<]*)<\/summary>/)),
|
||||
language: grab(/<language>([^<]*)<\/language>/),
|
||||
name,
|
||||
flavour: grab(/<flavour>([^<]*)<\/flavour>/),
|
||||
category: grab(/<category>([^<]*)<\/category>/),
|
||||
tags,
|
||||
mediaCount: Number(grab(/<mediaCount>([^<]*)<\/mediaCount>/)) || 0,
|
||||
author: grab(/<author>\s*<name>([^<]*)<\/name>\s*<\/author>/m),
|
||||
publisher: grab(/<publisher>\s*<name>([^<]*)<\/name>\s*<\/publisher>/m),
|
||||
articleCount: Number(grab(/<articleCount>([^<]*)<\/articleCount>/)) || 0,
|
||||
sizeMb: linkMatch ? +(Number((b.match(/length=["'](\d+)["']/) || [])[1] || 0) / 1024 / 1024).toFixed(1) : '?',
|
||||
href: linkMatch ? linkMatch[1] : null,
|
||||
sizeMb: href
|
||||
? +(Number((b.match(/length=["'](\d+)["']/) || [])[1] || 0) / 1024 / 1024).toFixed(1)
|
||||
: '?',
|
||||
href,
|
||||
icon: iconMatch ? new URL(iconMatch[1], catalog).href : null,
|
||||
viewer: name ? new URL(`viewer#${name}`, catalog).href : null,
|
||||
viewer: viewerName ? `${VIEWER_URL}/viewer#${viewerName}` : null,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
async function fetchEntries(term, lang, url = CATALOG_URL) {
|
||||
const params = new URLSearchParams({q: term, count: String(PAGE_SIZE), lang: lang || 'eng'});
|
||||
const entries = [];
|
||||
let start = 0;
|
||||
let total = Infinity;
|
||||
|
||||
while (start < total) {
|
||||
const params = new URLSearchParams({
|
||||
q: term,
|
||||
count: String(PAGE_SIZE),
|
||||
start: String(start),
|
||||
lang: lang || 'eng',
|
||||
});
|
||||
|
||||
const res = await fetch(`${url}/catalog/v2/entries?${params}`);
|
||||
if (!res.ok) throw new Error(`${res.status} ${res.statusText}`);
|
||||
return parseEntries(await res.text());
|
||||
|
||||
const xml = await res.text();
|
||||
const page = parseEntries(xml, url);
|
||||
const totalResults = Number((xml.match(/<totalResults>(\d+)<\/totalResults>/) || [])[1]);
|
||||
|
||||
entries.push(...page);
|
||||
|
||||
if (Number.isFinite(totalResults)) total = totalResults;
|
||||
if (!page.length || start + page.length >= total) break;
|
||||
|
||||
start += page.length;
|
||||
}
|
||||
|
||||
return entries;
|
||||
}
|
||||
|
||||
/** Looks up a single catalog entry by exact `name` (used to check for available updates). */
|
||||
@@ -48,30 +85,73 @@ export async function zimCatalogInfo(name, url = CATALOG_URL) {
|
||||
const params = new URLSearchParams({name, count: '5'});
|
||||
const res = await fetch(`${url}/catalog/v2/entries?${params}`);
|
||||
if (!res.ok) return null;
|
||||
return parseEntries(await res.text()).find(e => e.name === name) || null;
|
||||
|
||||
return parseEntries(await res.text(), url).find(e => e.name === name) || null;
|
||||
}
|
||||
|
||||
/** Searches the Kiwix catalog for ZIMs matching `terms` within `category`, ranked by term coverage then fuzzy similarity. */
|
||||
/** Searches the Kiwix catalog for ZIMs matching `terms`, ranked by relevance. */
|
||||
export async function zimCatalog(terms, opts = {lang: 'eng', count: 20, url: CATALOG_URL}) {
|
||||
opts = Object.assign({lang: 'eng', count: 20, url: CATALOG_URL}, opts)
|
||||
const termList = [...String(terms).split(',')].filter(Boolean).map(t => t.trim().toLowerCase());
|
||||
const results = await Promise.allSettled(termList.map(t => fetchEntries(t, opts.lang, opts.url)));
|
||||
const byName = new Map();
|
||||
opts = Object.assign({lang: 'eng', count: 20, url: CATALOG_URL}, opts);
|
||||
|
||||
const termList = [...String(terms).split(',')]
|
||||
.filter(Boolean)
|
||||
.map(t => t.trim().toLowerCase());
|
||||
|
||||
const results = await Promise.allSettled(
|
||||
termList.map(t => fetchEntries(t, opts.lang, opts.url))
|
||||
);
|
||||
|
||||
const byZim = new Map();
|
||||
|
||||
results.forEach((r, i) => {
|
||||
if (r.status !== 'fulfilled') return;
|
||||
|
||||
const term = termList[i];
|
||||
|
||||
for (const entry of r.value) {
|
||||
if (!entry.name) continue;
|
||||
if (!byName.has(entry.name)) byName.set(entry.name, {entry, hitTerms: new Set()});
|
||||
byName.get(entry.name).hitTerms.add(term);
|
||||
const zimId = entry.href || entry.id || entry.name;
|
||||
if (!zimId) continue;
|
||||
|
||||
if (!byZim.has(zimId)) {
|
||||
byZim.set(zimId, {entry, hitTerms: new Set()});
|
||||
}
|
||||
|
||||
byZim.get(zimId).hitTerms.add(term);
|
||||
}
|
||||
});
|
||||
if (!byName.size) return [];
|
||||
|
||||
return [...byName.values()].map(({entry, hitTerms}) => {
|
||||
const text = `${entry.title} ${entry.summary}`.trim();
|
||||
return {entry, hits: hitTerms.size, fuzzy: fuzzyMatch(text, ...termList).max};
|
||||
}).toSorted((a, b) =>
|
||||
b.hits - a.hits || b.fuzzy - a.fuzzy || b.entry.articleCount - a.entry.articleCount
|
||||
).slice(0, opts.count).map(r => r.entry);
|
||||
if (!byZim.size) return [];
|
||||
|
||||
return [...byZim.values()]
|
||||
.map(({entry, hitTerms}) => {
|
||||
const title = entry.title.toLowerCase();
|
||||
const summary = entry.summary.toLowerCase();
|
||||
const titleWords = title.split(/\W+/).filter(Boolean);
|
||||
|
||||
const exactTitle = termList.some(term => title === term);
|
||||
const exactWord = termList.some(term => titleWords.includes(term));
|
||||
const titleMatch = termList.some(term => title.includes(term));
|
||||
const titleFuzzy = fuzzyMatch(title, ...termList).max;
|
||||
const summaryFuzzy = fuzzyMatch(summary, ...termList).max;
|
||||
|
||||
return {
|
||||
entry,
|
||||
hits: hitTerms.size,
|
||||
exactTitle,
|
||||
exactWord,
|
||||
titleMatch,
|
||||
titleFuzzy,
|
||||
summaryFuzzy,
|
||||
};
|
||||
})
|
||||
.toSorted((a, b) =>
|
||||
Number(b.exactTitle) - Number(a.exactTitle)
|
||||
|| Number(b.exactWord) - Number(a.exactWord)
|
||||
|| Number(b.titleMatch) - Number(a.titleMatch)
|
||||
|| b.titleFuzzy - a.titleFuzzy
|
||||
|| b.summaryFuzzy - a.summaryFuzzy
|
||||
|| b.hits - a.hits
|
||||
|| new Date(b.entry.updated) - new Date(a.entry.updated)
|
||||
)
|
||||
.map(r => r.entry);
|
||||
}
|
||||
|
||||
+1
-1
@@ -107,7 +107,7 @@ export class ZimManager {
|
||||
const file = match.href + (match.href.endsWith('.zim') ? '' : '.zim');
|
||||
await fs.promises.rm(path.join(this.#dir, file), {force: true});
|
||||
await this.#ensureServer();
|
||||
await this.#waitUntilVicsible(file, {expect: false});
|
||||
await this.#waitUntilVisible(file, {expect: false});
|
||||
return {file: match.file, status: 'deleted'};
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -101,7 +101,7 @@ export function rerank(hits, termList) {
|
||||
const title = fieldScore(hit.title, termList, {fuzzy: true});
|
||||
const summary = fieldScore(hit.summary, termList);
|
||||
const titleLower = (hit.title || '').toLowerCase();
|
||||
const summaryLower = (hit.summary || '').toLowerCase();
|
||||
const summaryLower = (hit.summary?._text || '').toLowerCase();
|
||||
const phrase = termList.join(' ').toLowerCase();
|
||||
|
||||
const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0;
|
||||
|
||||
+17
-24
@@ -212,7 +212,7 @@ export class KiwixServer {
|
||||
|
||||
return makeArray(entries?.library?.book || []).map(e => {
|
||||
const tags = e.tags.split(';');
|
||||
const name = e.path.replace('.zim', '');
|
||||
const name = e.path.replace(/\.zim$/i, '');
|
||||
|
||||
return {
|
||||
id: e.id,
|
||||
@@ -262,53 +262,43 @@ export class KiwixServer {
|
||||
}
|
||||
|
||||
async #rawSearch(termList, scoped, bookMap, limit) {
|
||||
const groups = new Map();
|
||||
|
||||
for(const book of scoped) {
|
||||
const lang = book.language || '';
|
||||
(groups.get(lang) ?? groups.set(lang, []).get(lang)).push(book);
|
||||
}
|
||||
|
||||
const perGroup = await Promise.all([...groups.values()].map(async group => {
|
||||
const perBook = await Promise.all(scoped.map(async book => {
|
||||
const params = new URLSearchParams({
|
||||
pattern: termList.join(' '),
|
||||
format: 'xml',
|
||||
pageLength: String(limit),
|
||||
'books.name': book.href,
|
||||
});
|
||||
|
||||
for(const book of group)
|
||||
params.append('books.name', book.name);
|
||||
|
||||
const res = await fetch(`${this.baseUrl}/search?${params}`);
|
||||
if(!res.ok) return [];
|
||||
|
||||
const found = fromXml(await res.text())?.rss?.channel?.item || [];
|
||||
|
||||
return found.map(hit => {
|
||||
const book = bookMap.get(hit.book.title);
|
||||
if(!book) return null;
|
||||
return makeArray(found).map(hit => {
|
||||
const resultBook = bookMap.get(hit.book?.title) || book;
|
||||
|
||||
const prefix = `/content/${book.href}/`;
|
||||
const prefix = `/content/${resultBook.href}/`;
|
||||
const page = hit.link.startsWith(prefix)
|
||||
? hit.link.slice(prefix.length)
|
||||
: hit.link.replace(/^\/+/, '');
|
||||
|
||||
return {
|
||||
id: book.id,
|
||||
id: resultBook.id,
|
||||
title: hit.title,
|
||||
page,
|
||||
name: book.name,
|
||||
publisher: book.publisher,
|
||||
href: `${book.href}/${page}`,
|
||||
icon: book.icon,
|
||||
name: resultBook.name,
|
||||
publisher: resultBook.publisher,
|
||||
href: `${resultBook.href}/${page}`,
|
||||
icon: resultBook.icon,
|
||||
viewer: this.baseUrl + hit.link,
|
||||
summary: hit.description,
|
||||
xapianScore: +hit.score || 0,
|
||||
};
|
||||
}).filter(Boolean).sort((a, b) => b.xapianScore - a.xapianScore);
|
||||
}).filter(Boolean);
|
||||
}));
|
||||
|
||||
return rrfMerge(perGroup);
|
||||
return rrfMerge(perBook);
|
||||
}
|
||||
|
||||
/** Fulltext search across local ZIMs with ranking, diversification and spelling correction. */
|
||||
@@ -320,6 +310,7 @@ export class KiwixServer {
|
||||
|
||||
const books = await this.list();
|
||||
const bookMap = new Map(books.map(b => [b.title, b]));
|
||||
|
||||
const scoped = sources?.length
|
||||
? books.filter(b => sources.includes(b.name) || sources.includes(b.href))
|
||||
: books;
|
||||
@@ -360,8 +351,10 @@ export class KiwixServer {
|
||||
}
|
||||
}
|
||||
|
||||
const ranked = rerank(hits, activeTerms);
|
||||
|
||||
return {
|
||||
results: diversify(rerank(hits, activeTerms), limit),
|
||||
results: diversify(ranked, limit),
|
||||
spellcheck,
|
||||
};
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user