generated from ztimson/template
use kiwix-serach as prefilter before fuzzy ranking instead of building index
This commit is contained in:
116
src/reader.js
116
src/reader.js
@@ -1,11 +1,9 @@
|
||||
'use strict';
|
||||
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import {decompressPool} from './decompress.js';
|
||||
import {fuzzyMatch, titleFromUrl} from './utils.js';
|
||||
import {fuzzyMatch, titleFromUrl, kiwixSearch} from './utils.js';
|
||||
|
||||
const INDEX_VERSION = 1;
|
||||
const HEADER_SIZE = 80;
|
||||
const NS_CONTENT = 'C';
|
||||
const NS_METADATA = 'M';
|
||||
@@ -20,7 +18,6 @@ export class ZimReader {
|
||||
#header = null;
|
||||
#mimeTypes = [];
|
||||
#hasTitleListing = false;
|
||||
#index;
|
||||
#clusterCache = new Map(); // clusterNumber -> {data, extended, timer}
|
||||
#pending = new Map(); // clusterNumber -> Promise, dedupes concurrent misses
|
||||
#clusterCacheMax;
|
||||
@@ -36,15 +33,6 @@ export class ZimReader {
|
||||
this.#clusterTTL = clusterTTL;
|
||||
}
|
||||
|
||||
/** Full O(n) scan over the URL pointer list, used when there's no title index. */
|
||||
async #allDirents() {
|
||||
const dirents = [];
|
||||
for (let i = 0; i < this.#header.articleCount; i++) {
|
||||
dirents.push(await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, i)));
|
||||
}
|
||||
return dirents;
|
||||
}
|
||||
|
||||
/** Binary search the URL pointer list for namespace+url. */
|
||||
async #findByUrl(url, namespace) {
|
||||
const key = namespace + url;
|
||||
@@ -60,6 +48,22 @@ export class ZimReader {
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Binary search the title index for an exact title match. */
|
||||
async #findByTitle(title) {
|
||||
if (!this.#hasTitleListing) return null;
|
||||
const q = title.toLowerCase();
|
||||
let lo = 0, hi = this.#header.articleCount - 1;
|
||||
while (lo <= hi) {
|
||||
const mid = (lo + hi) >> 1;
|
||||
const urlIdx = (await this.#read(this.#header.titlePtrPos + mid * 4, 4)).readUInt32LE(0);
|
||||
const dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx));
|
||||
const t = dirent.title.toLowerCase();
|
||||
if (t === q) return dirent;
|
||||
if (t < q) lo = mid + 1; else hi = mid - 1;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Resets a cluster's idle-eviction timer. No-op when TTL disabled. */
|
||||
#touch(clusterNumber, entry) {
|
||||
if (!this.#clusterTTL) return;
|
||||
@@ -185,68 +189,6 @@ export class ZimReader {
|
||||
return this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, dirent.redirectIndex));
|
||||
}
|
||||
|
||||
/** Narrows to dirents near the term's alphabetical position in the title index. */
|
||||
async #titleIndexCandidates(term) {
|
||||
const q = term.toLowerCase();
|
||||
let lo = 0, hi = this.#header.articleCount - 1;
|
||||
while (lo < hi) {
|
||||
const mid = (lo + hi) >> 1;
|
||||
const urlIdx = (await this.#read(this.#header.titlePtrPos + mid * 4, 4)).readUInt32LE(0);
|
||||
const dirent = await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx));
|
||||
if (dirent.title.toLowerCase() < q) lo = mid + 1; else hi = mid;
|
||||
}
|
||||
// Widen around the prefix match since fuzzy scoring isn't purely alphabetical.
|
||||
const start = Math.max(0, lo - 50), end = Math.min(this.#header.articleCount, lo + 200);
|
||||
const dirents = [];
|
||||
for (let i = start; i < end; i++) {
|
||||
const urlIdx = (await this.#read(this.#header.titlePtrPos + i * 4, 4)).readUInt32LE(0);
|
||||
dirents.push(await this.#readDirent(await this.#ptr64(this.#header.urlPtrPos, urlIdx)));
|
||||
}
|
||||
return dirents;
|
||||
}
|
||||
|
||||
/** Path of the cached word index, kept in an `index/` subdirectory next to the archive. */
|
||||
#indexPath() {
|
||||
return path.join(path.dirname(this.path), 'index', `${path.basename(this.path)}.idx.json`);
|
||||
}
|
||||
|
||||
/** Builds (or loads a cached) word -> entry index, avoiding a re-scan on every search. */
|
||||
async #wordIndex() {
|
||||
if (this.#index) return this.#index;
|
||||
const cachePath = this.#indexPath();
|
||||
const stat = await fs.promises.stat(this.path);
|
||||
|
||||
if (fs.existsSync(cachePath)) {
|
||||
try {
|
||||
const cached = JSON.parse(await fs.promises.readFile(cachePath, 'utf8'));
|
||||
// Rebuild if stale: index predates the archive's current mtime (e.g. a re-download).
|
||||
if (cached.version === INDEX_VERSION && cached.mtimeMs >= stat.mtimeMs) {
|
||||
return (this.#index = cached);
|
||||
}
|
||||
} catch {}
|
||||
}
|
||||
|
||||
// Expensive full scan — only happens once per archive (or after it changes).
|
||||
const dirents = await this.#allDirents();
|
||||
const entries = [];
|
||||
const words = new Map(); // word -> [entryIndex, ...]
|
||||
for (const d of dirents) {
|
||||
if (d.namespace !== NS_CONTENT) continue;
|
||||
const idx = entries.length;
|
||||
entries.push({url: d.url, title: d.title, mimetype: d.mimetype});
|
||||
const text = `${d.title} ${titleFromUrl(d.url)}`.toLowerCase();
|
||||
for (const w of text.split(/\W+/).filter(Boolean)) {
|
||||
if (!words.has(w)) words.set(w, []);
|
||||
words.get(w).push(idx);
|
||||
}
|
||||
}
|
||||
|
||||
const index = {version: INDEX_VERSION, size: stat.size, mtimeMs: stat.mtimeMs, entries, words: Object.fromEntries(words)};
|
||||
await fs.promises.mkdir(path.dirname(cachePath), {recursive: true});
|
||||
await fs.promises.writeFile(cachePath, JSON.stringify(index));
|
||||
return (this.#index = index);
|
||||
}
|
||||
|
||||
async close() {
|
||||
if (this.#fd) await this.#fd.close();
|
||||
this.#fd = null;
|
||||
@@ -255,12 +197,10 @@ export class ZimReader {
|
||||
this.#pending.clear();
|
||||
}
|
||||
|
||||
/** Deletes the zim archive and its cached index (if any). Safe to call on unopened readers. */
|
||||
/** Deletes the zim archive. Safe to call on unopened readers. */
|
||||
async delete() {
|
||||
await this.close();
|
||||
this.#index = null;
|
||||
await fs.promises.rm(this.path, {force: true});
|
||||
await fs.promises.rm(this.#indexPath(), {force: true});
|
||||
}
|
||||
|
||||
/** Reads an 'M' namespace metadata value (e.g. Name, Date, Title). Returns null if missing. */
|
||||
@@ -325,27 +265,15 @@ export class ZimReader {
|
||||
|
||||
/**
|
||||
* Fuzzy-ranked title search. Accepts comma-separated `terms` the same way the
|
||||
* catalog search does. Uses the sorted title index when present (binary search
|
||||
* narrows the candidate window); falls back to a full linear scan otherwise
|
||||
* (common on ZIM v6+/zimit-generated archives with no title index).
|
||||
* catalog search does. Uses kiwix-search's embedded fulltext index as a prefilter
|
||||
* to narrow candidates before fuzzy scoring.
|
||||
*/
|
||||
async search(terms, {limit = 20, htmlOnly = true} = {}) {
|
||||
const termList = String(terms).split(',').map(t => t.trim()).filter(Boolean);
|
||||
if (!termList.length) return [];
|
||||
|
||||
let candidates;
|
||||
if (this.#hasTitleListing && this.#header.articleCount > 5000) {
|
||||
candidates = await this.#titleIndexCandidates(termList[0]);
|
||||
} else {
|
||||
const {entries, words} = await this.#wordIndex();
|
||||
// Pull candidates from postings of any word that starts with (or contains) the search term.
|
||||
const q = termList[0].toLowerCase();
|
||||
const idxSet = new Set();
|
||||
for (const [word, postings] of Object.entries(words)) {
|
||||
if (word.includes(q) || q.includes(word)) postings.forEach(i => idxSet.add(i));
|
||||
}
|
||||
candidates = [...idxSet].map(i => ({...entries[i], namespace: NS_CONTENT}));
|
||||
}
|
||||
const titles = await kiwixSearch(this.path, termList.join(' '));
|
||||
const candidates = (await Promise.all(titles.map(t => this.#findByTitle(t)))).filter(Boolean);
|
||||
|
||||
const scored = [];
|
||||
for (const dirent of candidates) {
|
||||
|
||||
Reference in New Issue
Block a user