generated from ztimson/template
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5bc27fa632 | ||
|
|
943280f326 | ||
|
|
dc24f48a36 | ||
|
|
812e72a8cd | ||
|
|
a45d7bad8c |
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@ztimson/zim-utils",
|
||||
"version": "0.4.2",
|
||||
"version": "0.4.6",
|
||||
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
|
||||
"author": "Zak Timson",
|
||||
"license": "MIT",
|
||||
|
||||
+101
-21
@@ -2,19 +2,29 @@ import {fuzzyMatch} from './utils.js';
|
||||
import {decodeHtml} from '@ztimson/utils';
|
||||
|
||||
export const CATALOG_URL = 'https://library.kiwix.org';
|
||||
export const VIEWER_URL = 'https://browse.library.kiwix.org';
|
||||
|
||||
const PAGE_SIZE = 100;
|
||||
|
||||
export function parseEntries(xml, catalog = CATALOG_URL) {
|
||||
const blocks = xml.match(/<entry>[\s\S]*?<\/entry>/g) || [];
|
||||
|
||||
return blocks.map(b => {
|
||||
const grab = re => (b.match(re) || [])[1] || '';
|
||||
const linkMatch = b.match(/<link[^>]*type=["']application\/x-zim[^"']*["'][^>]*href=["']([^"']+)["']/);
|
||||
const iconMatch = b.match(/<link[^>]*rel=["']http:\/\/opds-spec\.org\/image\/thumbnail["'][^>]*href=["']([^"']+)["']/)
|
||||
|| b.match(/<link[^>]*type=["']image\/[^"']*["'][^>]*href=["']([^"']+)["']/);
|
||||
|
||||
let tags = grab(/<tags>([^<]*)<\/tags>/);
|
||||
if (tags) tags = tags.split(';');
|
||||
|
||||
const name = grab(/<name>([^<]*)<\/name>/);
|
||||
const href = linkMatch ? linkMatch[1] : null;
|
||||
const filename = href
|
||||
? decodeURIComponent(new URL(href).pathname.split('/').pop()).replace(/\.meta4$/i, '')
|
||||
: name;
|
||||
const viewerName = filename.replace(/\.zim$/i, '');
|
||||
|
||||
return {
|
||||
id: grab(/<id>([^<]*)<\/id>/),
|
||||
title: decodeHtml(grab(/<title>([^<]*)<\/title>/)),
|
||||
@@ -22,25 +32,52 @@ export function parseEntries(xml, catalog = CATALOG_URL) {
|
||||
summary: decodeHtml(grab(/<summary>([^<]*)<\/summary>/)),
|
||||
language: grab(/<language>([^<]*)<\/language>/),
|
||||
name,
|
||||
flavour: grab(/<flavour>([^<]*)<\/flavour>/),
|
||||
category: grab(/<category>([^<]*)<\/category>/),
|
||||
tags,
|
||||
mediaCount: Number(grab(/<mediaCount>([^<]*)<\/mediaCount>/)) || 0,
|
||||
author: grab(/<author>\s*<name>([^<]*)<\/name>\s*<\/author>/m),
|
||||
publisher: grab(/<publisher>\s*<name>([^<]*)<\/name>\s*<\/publisher>/m),
|
||||
articleCount: Number(grab(/<articleCount>([^<]*)<\/articleCount>/)) || 0,
|
||||
sizeMb: linkMatch ? +(Number((b.match(/length=["'](\d+)["']/) || [])[1] || 0) / 1024 / 1024).toFixed(1) : '?',
|
||||
href: linkMatch ? linkMatch[1] : null,
|
||||
sizeMb: href
|
||||
? +(Number((b.match(/length=["'](\d+)["']/) || [])[1] || 0) / 1024 / 1024).toFixed(1)
|
||||
: '?',
|
||||
href,
|
||||
icon: iconMatch ? new URL(iconMatch[1], catalog).href : null,
|
||||
viewer: name ? new URL(`viewer#${name}`, catalog).href : null,
|
||||
viewer: viewerName ? `${VIEWER_URL}/viewer#${viewerName}` : null,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
async function fetchEntries(term, lang, url = CATALOG_URL) {
|
||||
const params = new URLSearchParams({q: term, count: String(PAGE_SIZE), lang: lang || 'eng'});
|
||||
const entries = [];
|
||||
let start = 0;
|
||||
let total = Infinity;
|
||||
|
||||
while (start < total) {
|
||||
const params = new URLSearchParams({
|
||||
q: term,
|
||||
count: String(PAGE_SIZE),
|
||||
start: String(start),
|
||||
lang: lang || 'eng',
|
||||
});
|
||||
|
||||
const res = await fetch(`${url}/catalog/v2/entries?${params}`);
|
||||
if (!res.ok) throw new Error(`${res.status} ${res.statusText}`);
|
||||
return parseEntries(await res.text());
|
||||
|
||||
const xml = await res.text();
|
||||
const page = parseEntries(xml, url);
|
||||
const totalResults = Number((xml.match(/<totalResults>(\d+)<\/totalResults>/) || [])[1]);
|
||||
|
||||
entries.push(...page);
|
||||
|
||||
if (Number.isFinite(totalResults)) total = totalResults;
|
||||
if (!page.length || start + page.length >= total) break;
|
||||
|
||||
start += page.length;
|
||||
}
|
||||
|
||||
return entries;
|
||||
}
|
||||
|
||||
/** Looks up a single catalog entry by exact `name` (used to check for available updates). */
|
||||
@@ -48,30 +85,73 @@ export async function zimCatalogInfo(name, url = CATALOG_URL) {
|
||||
const params = new URLSearchParams({name, count: '5'});
|
||||
const res = await fetch(`${url}/catalog/v2/entries?${params}`);
|
||||
if (!res.ok) return null;
|
||||
return parseEntries(await res.text()).find(e => e.name === name) || null;
|
||||
|
||||
return parseEntries(await res.text(), url).find(e => e.name === name) || null;
|
||||
}
|
||||
|
||||
/** Searches the Kiwix catalog for ZIMs matching `terms` within `category`, ranked by term coverage then fuzzy similarity. */
|
||||
/** Searches the Kiwix catalog for ZIMs matching `terms`, ranked by relevance. */
|
||||
export async function zimCatalog(terms, opts = {lang: 'eng', count: 20, url: CATALOG_URL}) {
|
||||
opts = Object.assign({lang: 'eng', count: 20, url: CATALOG_URL}, opts)
|
||||
const termList = [...String(terms).split(',')].filter(Boolean).map(t => t.trim().toLowerCase());
|
||||
const results = await Promise.allSettled(termList.map(t => fetchEntries(t, opts.lang, opts.url)));
|
||||
const byName = new Map();
|
||||
opts = Object.assign({lang: 'eng', count: 20, url: CATALOG_URL}, opts);
|
||||
|
||||
const termList = [...String(terms).split(',')]
|
||||
.filter(Boolean)
|
||||
.map(t => t.trim().toLowerCase());
|
||||
|
||||
const results = await Promise.allSettled(
|
||||
termList.map(t => fetchEntries(t, opts.lang, opts.url))
|
||||
);
|
||||
|
||||
const byZim = new Map();
|
||||
|
||||
results.forEach((r, i) => {
|
||||
if (r.status !== 'fulfilled') return;
|
||||
|
||||
const term = termList[i];
|
||||
|
||||
for (const entry of r.value) {
|
||||
if (!entry.name) continue;
|
||||
if (!byName.has(entry.name)) byName.set(entry.name, {entry, hitTerms: new Set()});
|
||||
byName.get(entry.name).hitTerms.add(term);
|
||||
const zimId = entry.href || entry.id || entry.name;
|
||||
if (!zimId) continue;
|
||||
|
||||
if (!byZim.has(zimId)) {
|
||||
byZim.set(zimId, {entry, hitTerms: new Set()});
|
||||
}
|
||||
|
||||
byZim.get(zimId).hitTerms.add(term);
|
||||
}
|
||||
});
|
||||
if (!byName.size) return [];
|
||||
|
||||
return [...byName.values()].map(({entry, hitTerms}) => {
|
||||
const text = `${entry.title} ${entry.summary}`.trim();
|
||||
return {entry, hits: hitTerms.size, fuzzy: fuzzyMatch(text, ...termList).max};
|
||||
}).toSorted((a, b) =>
|
||||
b.hits - a.hits || b.fuzzy - a.fuzzy || b.entry.articleCount - a.entry.articleCount
|
||||
).slice(0, opts.count).map(r => r.entry);
|
||||
if (!byZim.size) return [];
|
||||
|
||||
return [...byZim.values()]
|
||||
.map(({entry, hitTerms}) => {
|
||||
const title = entry.title.toLowerCase();
|
||||
const summary = entry.summary.toLowerCase();
|
||||
const titleWords = title.split(/\W+/).filter(Boolean);
|
||||
|
||||
const exactTitle = termList.some(term => title === term);
|
||||
const exactWord = termList.some(term => titleWords.includes(term));
|
||||
const titleMatch = termList.some(term => title.includes(term));
|
||||
const titleFuzzy = fuzzyMatch(title, ...termList).max;
|
||||
const summaryFuzzy = fuzzyMatch(summary, ...termList).max;
|
||||
|
||||
return {
|
||||
entry,
|
||||
hits: hitTerms.size,
|
||||
exactTitle,
|
||||
exactWord,
|
||||
titleMatch,
|
||||
titleFuzzy,
|
||||
summaryFuzzy,
|
||||
};
|
||||
})
|
||||
.toSorted((a, b) =>
|
||||
Number(b.exactTitle) - Number(a.exactTitle)
|
||||
|| Number(b.exactWord) - Number(a.exactWord)
|
||||
|| Number(b.titleMatch) - Number(a.titleMatch)
|
||||
|| b.titleFuzzy - a.titleFuzzy
|
||||
|| b.summaryFuzzy - a.summaryFuzzy
|
||||
|| b.hits - a.hits
|
||||
|| new Date(b.entry.updated) - new Date(a.entry.updated)
|
||||
)
|
||||
.map(r => r.entry);
|
||||
}
|
||||
|
||||
+9
-3
@@ -32,6 +32,12 @@ function tokenize(terms) {
|
||||
return String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
|
||||
}
|
||||
|
||||
function normalizeSummary(summary) {
|
||||
if(typeof summary === 'string') return summary;
|
||||
if(summary && typeof summary === 'object') return summary._text || '';
|
||||
return '';
|
||||
}
|
||||
|
||||
export class KiwixServer {
|
||||
static #empty = '<?xml version="1.0" encoding="UTF-8" ?>\n<library version="20110515"></library>\n';
|
||||
|
||||
@@ -218,7 +224,7 @@ export class KiwixServer {
|
||||
id: e.id,
|
||||
title: e.title,
|
||||
updated: new Date(e.date),
|
||||
summary: e.description,
|
||||
summary: normalizeSummary(e.description),
|
||||
language: e.language,
|
||||
name: e.name,
|
||||
category: tags.find(t => t.startsWith('_category'))?.slice(10) || '',
|
||||
@@ -275,7 +281,7 @@ export class KiwixServer {
|
||||
|
||||
const found = fromXml(await res.text())?.rss?.channel?.item || [];
|
||||
|
||||
return found.map(hit => {
|
||||
return makeArray(found).map(hit => {
|
||||
const resultBook = bookMap.get(hit.book?.title) || book;
|
||||
|
||||
const prefix = `/content/${resultBook.href}/`;
|
||||
@@ -292,7 +298,7 @@ export class KiwixServer {
|
||||
href: `${resultBook.href}/${page}`,
|
||||
icon: resultBook.icon,
|
||||
viewer: this.baseUrl + hit.link,
|
||||
summary: hit.description,
|
||||
summary: normalizeSummary(hit.description),
|
||||
xapianScore: +hit.score || 0,
|
||||
};
|
||||
}).filter(Boolean);
|
||||
|
||||
Reference in New Issue
Block a user