From 812e72a8cde72527f8acc617c972e832251ef505 Mon Sep 17 00:00:00 2001 From: ztimson Date: Sun, 20 Sep 2026 23:26:57 -0400 Subject: [PATCH] Better catalog search --- src/catalog.js | 128 +++++++++++++++++++++++++++++++++++++++---------- 1 file changed, 104 insertions(+), 24 deletions(-) diff --git a/src/catalog.js b/src/catalog.js index 45249e5..f5bba0c 100644 --- a/src/catalog.js +++ b/src/catalog.js @@ -2,19 +2,29 @@ import {fuzzyMatch} from './utils.js'; import {decodeHtml} from '@ztimson/utils'; export const CATALOG_URL = 'https://library.kiwix.org'; +export const VIEWER_URL = 'https://browse.library.kiwix.org'; const PAGE_SIZE = 100; export function parseEntries(xml, catalog = CATALOG_URL) { const blocks = xml.match(/[\s\S]*?<\/entry>/g) || []; + return blocks.map(b => { const grab = re => (b.match(re) || [])[1] || ''; const linkMatch = b.match(/]*type=["']application\/x-zim[^"']*["'][^>]*href=["']([^"']+)["']/); const iconMatch = b.match(/]*rel=["']http:\/\/opds-spec\.org\/image\/thumbnail["'][^>]*href=["']([^"']+)["']/) || b.match(/]*type=["']image\/[^"']*["'][^>]*href=["']([^"']+)["']/); + let tags = grab(/([^<]*)<\/tags>/); - if(tags) tags = tags.split(';'); + if (tags) tags = tags.split(';'); + const name = grab(/([^<]*)<\/name>/); + const href = linkMatch ? linkMatch[1] : null; + const filename = href + ? decodeURIComponent(new URL(href).pathname.split('/').pop()).replace(/\.meta4$/i, '') + : name; + const viewerName = filename.replace(/\.zim$/i, ''); + return { id: grab(/([^<]*)<\/id>/), title: decodeHtml(grab(/([^<]*)<\/title>/)), @@ -22,25 +32,52 @@ export function parseEntries(xml, catalog = CATALOG_URL) { summary: decodeHtml(grab(/<summary>([^<]*)<\/summary>/)), language: grab(/<language>([^<]*)<\/language>/), name, + flavour: grab(/<flavour>([^<]*)<\/flavour>/), category: grab(/<category>([^<]*)<\/category>/), tags, mediaCount: Number(grab(/<mediaCount>([^<]*)<\/mediaCount>/)) || 0, author: grab(/<author>\s*<name>([^<]*)<\/name>\s*<\/author>/m), publisher: grab(/<publisher>\s*<name>([^<]*)<\/name>\s*<\/publisher>/m), articleCount: Number(grab(/<articleCount>([^<]*)<\/articleCount>/)) || 0, - sizeMb: linkMatch ? +(Number((b.match(/length=["'](\d+)["']/) || [])[1] || 0) / 1024 / 1024).toFixed(1) : '?', - href: linkMatch ? linkMatch[1] : null, + sizeMb: href + ? +(Number((b.match(/length=["'](\d+)["']/) || [])[1] || 0) / 1024 / 1024).toFixed(1) + : '?', + href, icon: iconMatch ? new URL(iconMatch[1], catalog).href : null, - viewer: name ? new URL(`viewer#${name}`, catalog).href : null, + viewer: viewerName ? `${VIEWER_URL}/viewer#${viewerName}` : null, }; }); } async function fetchEntries(term, lang, url = CATALOG_URL) { - const params = new URLSearchParams({q: term, count: String(PAGE_SIZE), lang: lang || 'eng'}); - const res = await fetch(`${url}/catalog/v2/entries?${params}`); - if (!res.ok) throw new Error(`${res.status} ${res.statusText}`); - return parseEntries(await res.text()); + const entries = []; + let start = 0; + let total = Infinity; + + while (start < total) { + const params = new URLSearchParams({ + q: term, + count: String(PAGE_SIZE), + start: String(start), + lang: lang || 'eng', + }); + + const res = await fetch(`${url}/catalog/v2/entries?${params}`); + if (!res.ok) throw new Error(`${res.status} ${res.statusText}`); + + const xml = await res.text(); + const page = parseEntries(xml, url); + const totalResults = Number((xml.match(/<totalResults>(\d+)<\/totalResults>/) || [])[1]); + + entries.push(...page); + + if (Number.isFinite(totalResults)) total = totalResults; + if (!page.length || start + page.length >= total) break; + + start += page.length; + } + + return entries; } /** Looks up a single catalog entry by exact `name` (used to check for available updates). */ @@ -48,30 +85,73 @@ export async function zimCatalogInfo(name, url = CATALOG_URL) { const params = new URLSearchParams({name, count: '5'}); const res = await fetch(`${url}/catalog/v2/entries?${params}`); if (!res.ok) return null; - return parseEntries(await res.text()).find(e => e.name === name) || null; + + return parseEntries(await res.text(), url).find(e => e.name === name) || null; } -/** Searches the Kiwix catalog for ZIMs matching `terms` within `category`, ranked by term coverage then fuzzy similarity. */ +/** Searches the Kiwix catalog for ZIMs matching `terms`, ranked by relevance. */ export async function zimCatalog(terms, opts = {lang: 'eng', count: 20, url: CATALOG_URL}) { - opts = Object.assign({lang: 'eng', count: 20, url: CATALOG_URL}, opts) - const termList = [...String(terms).split(',')].filter(Boolean).map(t => t.trim().toLowerCase()); - const results = await Promise.allSettled(termList.map(t => fetchEntries(t, opts.lang, opts.url))); - const byName = new Map(); + opts = Object.assign({lang: 'eng', count: 20, url: CATALOG_URL}, opts); + + const termList = [...String(terms).split(',')] + .filter(Boolean) + .map(t => t.trim().toLowerCase()); + + const results = await Promise.allSettled( + termList.map(t => fetchEntries(t, opts.lang, opts.url)) + ); + + const byZim = new Map(); + results.forEach((r, i) => { if (r.status !== 'fulfilled') return; + const term = termList[i]; + for (const entry of r.value) { - if (!entry.name) continue; - if (!byName.has(entry.name)) byName.set(entry.name, {entry, hitTerms: new Set()}); - byName.get(entry.name).hitTerms.add(term); + const zimId = entry.href || entry.id || entry.name; + if (!zimId) continue; + + if (!byZim.has(zimId)) { + byZim.set(zimId, {entry, hitTerms: new Set()}); + } + + byZim.get(zimId).hitTerms.add(term); } }); - if (!byName.size) return []; - return [...byName.values()].map(({entry, hitTerms}) => { - const text = `${entry.title} ${entry.summary}`.trim(); - return {entry, hits: hitTerms.size, fuzzy: fuzzyMatch(text, ...termList).max}; - }).toSorted((a, b) => - b.hits - a.hits || b.fuzzy - a.fuzzy || b.entry.articleCount - a.entry.articleCount - ).slice(0, opts.count).map(r => r.entry); + if (!byZim.size) return []; + + return [...byZim.values()] + .map(({entry, hitTerms}) => { + const title = entry.title.toLowerCase(); + const summary = entry.summary.toLowerCase(); + const titleWords = title.split(/\W+/).filter(Boolean); + + const exactTitle = termList.some(term => title === term); + const exactWord = termList.some(term => titleWords.includes(term)); + const titleMatch = termList.some(term => title.includes(term)); + const titleFuzzy = fuzzyMatch(title, ...termList).max; + const summaryFuzzy = fuzzyMatch(summary, ...termList).max; + + return { + entry, + hits: hitTerms.size, + exactTitle, + exactWord, + titleMatch, + titleFuzzy, + summaryFuzzy, + }; + }) + .toSorted((a, b) => + Number(b.exactTitle) - Number(a.exactTitle) + || Number(b.exactWord) - Number(a.exactWord) + || Number(b.titleMatch) - Number(a.titleMatch) + || b.titleFuzzy - a.titleFuzzy + || b.summaryFuzzy - a.summaryFuzzy + || b.hits - a.hits + || new Date(b.entry.updated) - new Date(a.entry.updated) + ) + .map(r => r.entry); }