generated from ztimson/template
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0a7e87f6b7 | ||
|
|
e12c3e3cc6 | ||
|
|
8d4258b951 | ||
|
|
15f15e06f1 | ||
|
|
997bd6064d | ||
|
|
4b86fe8051 |
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@ztimson/zim-utils",
|
"name": "@ztimson/zim-utils",
|
||||||
"version": "0.3.6",
|
"version": "0.4.2",
|
||||||
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
|
"description": "Native, dependency-light ZIM archive reader/searcher and Kiwix catalog downloader for Node.js",
|
||||||
"author": "Zak Timson",
|
"author": "Zak Timson",
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
|
|||||||
+34
-10
@@ -5,8 +5,13 @@ import {Readable} from 'node:stream';
|
|||||||
import {KiwixServer} from './server.js';
|
import {KiwixServer} from './server.js';
|
||||||
import {zimCatalog, zimCatalogInfo, CATALOG_URL} from './catalog.js';
|
import {zimCatalog, zimCatalogInfo, CATALOG_URL} from './catalog.js';
|
||||||
|
|
||||||
|
const VISIBLE_TIMEOUT = 5_000;
|
||||||
|
const VISIBLE_POLL_INTERVAL = 200;
|
||||||
|
|
||||||
/** Manages a local directory of ZIM archives: catalog search, downloads, update checks, deletion.
|
/** Manages a local directory of ZIM archives: catalog search, downloads, update checks, deletion.
|
||||||
* Reuses (or owns & lazily starts) a KiwixServer for local listing/search, so `list()` and `catalog()` return the same shape. */
|
* Reuses (or owns & lazily starts) a KiwixServer for local listing/search, so `list()` and `catalog()` return the same shape.
|
||||||
|
* Doesn't drive library.xml/kiwix-serve reloads itself - the owning KiwixServer watches its directory and
|
||||||
|
* reloads automatically whenever .zim files change, no matter which ZimManager (local or remote-attached) wrote them. */
|
||||||
export class ZimManager {
|
export class ZimManager {
|
||||||
#catalogUrl;
|
#catalogUrl;
|
||||||
#dir;
|
#dir;
|
||||||
@@ -24,6 +29,10 @@ export class ZimManager {
|
|||||||
this.#server = server ?? new KiwixServer(dir, {port, host, binDir, url});
|
this.#server = server ?? new KiwixServer(dir, {port, host, binDir, url});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#stripDate(filename) {
|
||||||
|
return filename.replace(/\.zim$/i, '').replace(/_\d{4}-\d{2}(?:_\d+)?$/, '');
|
||||||
|
}
|
||||||
|
|
||||||
/** The KiwixServer backing this manager - reuse it directly for content/search access, or pass into another ZimManager. */
|
/** The KiwixServer backing this manager - reuse it directly for content/search access, or pass into another ZimManager. */
|
||||||
get server() { return this.#server; }
|
get server() { return this.#server; }
|
||||||
|
|
||||||
@@ -52,6 +61,20 @@ export class ZimManager {
|
|||||||
return {res, url: m[1]};
|
return {res, url: m[1]};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Polls list() until `filename` shows up (or disappears, if `expect: false`), so callers get an accurate
|
||||||
|
* status back rather than one that's ahead of what the server has actually picked up yet. The owning
|
||||||
|
* KiwixServer's directory watcher does the real reload work in the background; this just waits for it. */
|
||||||
|
async #waitUntilVisible(filename, {expect = true, timeout = VISIBLE_TIMEOUT} = {}) {
|
||||||
|
const deadline = Date.now() + timeout;
|
||||||
|
while (Date.now() < deadline) {
|
||||||
|
const local = await this.list();
|
||||||
|
const present = local.some(l => l.file === filename);
|
||||||
|
if (present === expect) return true;
|
||||||
|
await new Promise(r => setTimeout(r, VISIBLE_POLL_INTERVAL));
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
async #update(name, catalogEntry, localMatch, force) {
|
async #update(name, catalogEntry, localMatch, force) {
|
||||||
const remoteDate = catalogEntry.updated ? new Date(catalogEntry.updated) : null;
|
const remoteDate = catalogEntry.updated ? new Date(catalogEntry.updated) : null;
|
||||||
const localDate = localMatch?.updated ?? null;
|
const localDate = localMatch?.updated ?? null;
|
||||||
@@ -63,8 +86,8 @@ export class ZimManager {
|
|||||||
const destPath = path.join(this.#dir, filename);
|
const destPath = path.join(this.#dir, filename);
|
||||||
await this.#download(catalogEntry.href, destPath);
|
await this.#download(catalogEntry.href, destPath);
|
||||||
if (localMatch && localMatch.file !== filename) await fs.promises.rm(path.join(this.#dir, localMatch.file), {force: true});
|
if (localMatch && localMatch.file !== filename) await fs.promises.rm(path.join(this.#dir, localMatch.file), {force: true});
|
||||||
const server = await this.#ensureServer();
|
await this.#ensureServer();
|
||||||
await server.reload();
|
await this.#waitUntilVisible(filename, {expect: true});
|
||||||
return {name, status: 'updated', file: filename};
|
return {name, status: 'updated', file: filename};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -83,8 +106,8 @@ export class ZimManager {
|
|||||||
if (!match) throw new Error(`ZIM not found locally: ${nameOrFile}`);
|
if (!match) throw new Error(`ZIM not found locally: ${nameOrFile}`);
|
||||||
const file = match.href + (match.href.endsWith('.zim') ? '' : '.zim');
|
const file = match.href + (match.href.endsWith('.zim') ? '' : '.zim');
|
||||||
await fs.promises.rm(path.join(this.#dir, file), {force: true});
|
await fs.promises.rm(path.join(this.#dir, file), {force: true});
|
||||||
const server = await this.#ensureServer();
|
await this.#ensureServer();
|
||||||
await server.reload();
|
await this.#waitUntilVisible(file, {expect: false});
|
||||||
return {file: match.file, status: 'deleted'};
|
return {file: match.file, status: 'deleted'};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -93,11 +116,12 @@ export class ZimManager {
|
|||||||
const {url: finalUrl} = await this.#resolveUrl(href);
|
const {url: finalUrl} = await this.#resolveUrl(href);
|
||||||
const filename = path.basename(new URL(finalUrl).pathname).replace(/\.meta4$/i, '');
|
const filename = path.basename(new URL(finalUrl).pathname).replace(/\.meta4$/i, '');
|
||||||
await fs.promises.mkdir(this.#dir, {recursive: true});
|
await fs.promises.mkdir(this.#dir, {recursive: true});
|
||||||
const name = filename.replace(/\.zim$/i, '').replace(/_\d{4}-\d{2}(?:_\d+)?$/, '');
|
|
||||||
|
|
||||||
const local = await this.list();
|
const local = await this.list();
|
||||||
|
const exact = local.find(l => l.file === filename);
|
||||||
|
if (exact && !force) return {name: exact.name ?? this.#stripDate(filename), status: 'skipped', reason: 'already downloaded', file: filename};
|
||||||
|
const name = this.#stripDate(filename);
|
||||||
const localMatch = local.find(l => l.name === name) ?? null;
|
const localMatch = local.find(l => l.name === name) ?? null;
|
||||||
const catalogEntry = await zimCatalogInfo(name, this.#catalogUrl) || {name, updated: null, href: href};
|
const catalogEntry = await zimCatalogInfo(name, this.#catalogUrl) || {name, updated: null, href};
|
||||||
return this.#update(name, catalogEntry, localMatch, force);
|
return this.#update(name, catalogEntry, localMatch, force);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -130,9 +154,9 @@ export class ZimManager {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/** Two-pass fulltext search across every local ZIM, returns enriched results matching catalog/list shape. */
|
/** Two-pass fulltext search across every local ZIM, returns enriched results matching catalog/list shape. */
|
||||||
async search(terms, limit = 20) {
|
async search(terms, {limit = 20, sources = null} = {}) {
|
||||||
const server = await this.#ensureServer();
|
const server = await this.#ensureServer();
|
||||||
return server.search(terms, limit);
|
return server.search(terms, {limit, sources});
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Checks all local ZIMs against the catalog and updates any that are outdated. */
|
/** Checks all local ZIMs against the catalog and updates any that are outdated. */
|
||||||
|
|||||||
+51
-4
@@ -6,6 +6,8 @@ import {decompressPool} from './decompress.js';
|
|||||||
const HEADER_SIZE = 80;
|
const HEADER_SIZE = 80;
|
||||||
const NS_CONTENT = 'C';
|
const NS_CONTENT = 'C';
|
||||||
const NS_METADATA = 'M';
|
const NS_METADATA = 'M';
|
||||||
|
const VOCAB_CACHE_LIMIT = 4;
|
||||||
|
const vocabCache = new Map(); // zimPath -> {mtimeMs, size, words: Set<string>};
|
||||||
|
|
||||||
async function readAt(fd, pos, length) {
|
async function readAt(fd, pos, length) {
|
||||||
const buf = Buffer.alloc(length);
|
const buf = Buffer.alloc(length);
|
||||||
@@ -33,7 +35,10 @@ async function readMimeTypes(fd, mimeListPos) {
|
|||||||
for (;;) {
|
for (;;) {
|
||||||
str += (await readAt(fd, pos, 1024)).toString('binary');
|
str += (await readAt(fd, pos, 1024)).toString('binary');
|
||||||
const end = str.indexOf('\0\0');
|
const end = str.indexOf('\0\0');
|
||||||
if (end !== -1) { str = str.slice(0, end + 1); break; }
|
if (end !== -1) {
|
||||||
|
str = str.slice(0, end + 1);
|
||||||
|
break;
|
||||||
|
}
|
||||||
pos += 1024;
|
pos += 1024;
|
||||||
}
|
}
|
||||||
return str.split('\0').filter(Boolean);
|
return str.split('\0').filter(Boolean);
|
||||||
@@ -50,8 +55,15 @@ async function readDirent(fd, offset) {
|
|||||||
o += 4; // revision, unused
|
o += 4; // revision, unused
|
||||||
|
|
||||||
let redirectIndex = null, cluster = null, blob = null;
|
let redirectIndex = null, cluster = null, blob = null;
|
||||||
if (mimetype === 0xffff) { redirectIndex = buf.readUInt32LE(o); o += 4; }
|
if (mimetype === 0xffff) {
|
||||||
else { cluster = buf.readUInt32LE(o); o += 4; blob = buf.readUInt32LE(o); o += 4; }
|
redirectIndex = buf.readUInt32LE(o);
|
||||||
|
o += 4;
|
||||||
|
} else {
|
||||||
|
cluster = buf.readUInt32LE(o);
|
||||||
|
o += 4;
|
||||||
|
blob = buf.readUInt32LE(o);
|
||||||
|
o += 4;
|
||||||
|
}
|
||||||
|
|
||||||
const urlEnd = buf.indexOf(0, o);
|
const urlEnd = buf.indexOf(0, o);
|
||||||
if (urlEnd === -1) continue;
|
if (urlEnd === -1) continue;
|
||||||
@@ -73,7 +85,8 @@ async function findByUrl(fd, header, url, namespace) {
|
|||||||
const dirKey = dirent.namespace + dirent.url;
|
const dirKey = dirent.namespace + dirent.url;
|
||||||
const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0;
|
const cmp = key < dirKey ? -1 : key > dirKey ? 1 : 0;
|
||||||
if (cmp === 0) return dirent;
|
if (cmp === 0) return dirent;
|
||||||
if (cmp < 0) hi = mid - 1; else lo = mid + 1;
|
if (cmp < 0) hi = mid - 1;
|
||||||
|
else lo = mid + 1;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
@@ -136,3 +149,37 @@ export async function readZimMetadata(zimPath) {
|
|||||||
author: creator, publisher,
|
author: creator, publisher,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Unique, lowercased words pulled from every article title in a ZIM.
|
||||||
|
* Only the deduped Set is cached (not the raw title list), and only for the
|
||||||
|
* last VOCAB_CACHE_LIMIT ZIMs touched (evicted least-recently-used), so memory scales with recent search activity rather than total library size.
|
||||||
|
* First call per ZIM does a full O(articleCount) directory walk; cached after that until the file's mtime/size changes.
|
||||||
|
*/
|
||||||
|
export async function titleVocabulary(zimPath) {
|
||||||
|
const stat = await fs.promises.stat(zimPath);
|
||||||
|
const cached = vocabCache.get(zimPath);
|
||||||
|
if (cached && cached.mtimeMs === stat.mtimeMs && cached.size === stat.size) {
|
||||||
|
vocabCache.delete(zimPath);
|
||||||
|
vocabCache.set(zimPath, cached); // bump to most-recently-used
|
||||||
|
return cached.words;
|
||||||
|
}
|
||||||
|
|
||||||
|
const fd = await fs.promises.open(zimPath, 'r');
|
||||||
|
let words;
|
||||||
|
try {
|
||||||
|
const header = await readHeader(fd);
|
||||||
|
words = new Set();
|
||||||
|
for (let i = 0; i < header.articleCount; i++) {
|
||||||
|
const dirent = await readDirent(fd, await ptr64(fd, header.urlPtrPos, i));
|
||||||
|
if (dirent.namespace !== NS_CONTENT || dirent.mimetype === 0xffff) continue; // skip redirects/non-content
|
||||||
|
for (const w of (dirent.title || dirent.url).toLowerCase().split(/\W+/)) if (w) words.add(w);
|
||||||
|
}
|
||||||
|
} finally {
|
||||||
|
await fd.close();
|
||||||
|
}
|
||||||
|
|
||||||
|
vocabCache.set(zimPath, {mtimeMs: stat.mtimeMs, size: stat.size, words});
|
||||||
|
if (vocabCache.size > VOCAB_CACHE_LIMIT) vocabCache.delete(vocabCache.keys().next().value);
|
||||||
|
return words;
|
||||||
|
}
|
||||||
|
|||||||
+225
@@ -0,0 +1,225 @@
|
|||||||
|
'use strict';
|
||||||
|
|
||||||
|
import {fuzzyMatch} from './utils.js';
|
||||||
|
|
||||||
|
const RRF_K = 60;
|
||||||
|
|
||||||
|
/** Merges independently-ranked result lists without comparing their raw Xapian scores. */
|
||||||
|
export function rrfMerge(groups) {
|
||||||
|
const merged = new Map();
|
||||||
|
|
||||||
|
for(const group of groups) {
|
||||||
|
group.forEach((hit, rank) => {
|
||||||
|
const contribution = 1 / (RRF_K + rank + 1);
|
||||||
|
const entry = merged.get(hit.href);
|
||||||
|
|
||||||
|
if(entry) entry.score += contribution;
|
||||||
|
else merged.set(hit.href, {hit, score: contribution});
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
return [...merged.values()]
|
||||||
|
.sort((a, b) => b.score - a.score)
|
||||||
|
.map(({hit, score}) => ({...hit, rrf: score}));
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Calculates field relevance from exact terms, phrase matches, proximity,
|
||||||
|
* match density and optionally fuzzy similarity.
|
||||||
|
*/
|
||||||
|
function fieldScore(text, terms, {fuzzy = false} = {}) {
|
||||||
|
const value = String(text || '').trim().toLowerCase();
|
||||||
|
|
||||||
|
if(!value || !terms.length) {
|
||||||
|
return {
|
||||||
|
exact: 0,
|
||||||
|
coverage: 0,
|
||||||
|
proximity: 0,
|
||||||
|
density: 0,
|
||||||
|
fuzzy: 0,
|
||||||
|
phrase: 0,
|
||||||
|
score: 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
const words = value.split(/\W+/).filter(Boolean);
|
||||||
|
const normalizedTerms = terms.map(t => t.toLowerCase());
|
||||||
|
const phrase = normalizedTerms.join(' ');
|
||||||
|
|
||||||
|
const exactTerms = normalizedTerms.filter(t => words.includes(t));
|
||||||
|
const substringTerms = normalizedTerms.filter(t => value.includes(t));
|
||||||
|
|
||||||
|
const coverage = exactTerms.length / normalizedTerms.length;
|
||||||
|
const substringCoverage = substringTerms.length / normalizedTerms.length;
|
||||||
|
const exact = normalizedTerms.every(t => words.includes(t)) ? 1 : coverage;
|
||||||
|
const phraseScore = value.includes(phrase) ? 1 : 0;
|
||||||
|
|
||||||
|
let proximity = 0;
|
||||||
|
|
||||||
|
if(normalizedTerms.length > 1) {
|
||||||
|
const positions = [];
|
||||||
|
|
||||||
|
for(const term of normalizedTerms) {
|
||||||
|
const index = words.indexOf(term);
|
||||||
|
if(index !== -1) positions.push(index);
|
||||||
|
}
|
||||||
|
|
||||||
|
if(positions.length > 1) {
|
||||||
|
const span = Math.max(...positions) - Math.min(...positions);
|
||||||
|
proximity = 1 / Math.max(1, span);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const matchedChars = substringTerms.reduce((sum, term) => sum + term.length, 0);
|
||||||
|
const density = Math.min(1, matchedChars / Math.max(1, value.length * 0.25));
|
||||||
|
const fuzzyScore = fuzzy ? fuzzyMatch(value, ...normalizedTerms).avg : 0;
|
||||||
|
|
||||||
|
const score =
|
||||||
|
phraseScore * 1 +
|
||||||
|
exact * 0.8 +
|
||||||
|
proximity * 0.35 +
|
||||||
|
substringCoverage * 0.25 +
|
||||||
|
density * 0.15 +
|
||||||
|
fuzzyScore * 0.75;
|
||||||
|
|
||||||
|
return {
|
||||||
|
exact,
|
||||||
|
coverage,
|
||||||
|
proximity,
|
||||||
|
density,
|
||||||
|
fuzzy: fuzzyScore,
|
||||||
|
phrase: phraseScore,
|
||||||
|
score,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Reranks candidates using field-aware relevance while retaining RRF as the baseline. */
|
||||||
|
export function rerank(hits, termList) {
|
||||||
|
if(!termList.length) return hits;
|
||||||
|
|
||||||
|
return hits.map(hit => {
|
||||||
|
const title = fieldScore(hit.title, termList, {fuzzy: true});
|
||||||
|
const summary = fieldScore(hit.summary, termList);
|
||||||
|
const titleLower = (hit.title || '').toLowerCase();
|
||||||
|
const summaryLower = (hit.summary?._text || '').toLowerCase();
|
||||||
|
const phrase = termList.join(' ').toLowerCase();
|
||||||
|
|
||||||
|
const titleExactPhrase = titleLower.includes(phrase) ? 1 : 0;
|
||||||
|
const summaryExactPhrase = summaryLower.includes(phrase) ? 1 : 0;
|
||||||
|
|
||||||
|
const allTitleTerms = termList.every(term =>
|
||||||
|
titleLower.split(/\W+/).includes(term.toLowerCase())
|
||||||
|
) ? 1 : 0;
|
||||||
|
|
||||||
|
const allSummaryTerms = termList.every(term =>
|
||||||
|
summaryLower.includes(term.toLowerCase())
|
||||||
|
) ? 1 : 0;
|
||||||
|
|
||||||
|
const finalScore =
|
||||||
|
hit.rrf +
|
||||||
|
title.score * 1.25 +
|
||||||
|
summary.score * 0.35 +
|
||||||
|
titleExactPhrase * 1.5 +
|
||||||
|
allTitleTerms * 0.75 +
|
||||||
|
summaryExactPhrase * 0.2 +
|
||||||
|
allSummaryTerms * 0.15;
|
||||||
|
|
||||||
|
return {
|
||||||
|
...hit,
|
||||||
|
finalScore,
|
||||||
|
ranking: {
|
||||||
|
rrf: hit.rrf,
|
||||||
|
title: title.score,
|
||||||
|
summary: summary.score,
|
||||||
|
phrase: titleExactPhrase,
|
||||||
|
coverage: title.coverage,
|
||||||
|
proximity: title.proximity,
|
||||||
|
fuzzy: title.fuzzy,
|
||||||
|
},
|
||||||
|
};
|
||||||
|
}).sort((a, b) => b.finalScore - a.finalScore);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Round-robins results between source ZIMs so one archive cannot dominate the page. */
|
||||||
|
export function diversify(hits, limit) {
|
||||||
|
const byBook = new Map();
|
||||||
|
|
||||||
|
for(const hit of hits)
|
||||||
|
(byBook.get(hit.name) ?? byBook.set(hit.name, []).get(hit.name)).push(hit);
|
||||||
|
|
||||||
|
for(const list of byBook.values())
|
||||||
|
list.sort((a, b) => b.finalScore - a.finalScore);
|
||||||
|
|
||||||
|
const queues = [...byBook.values()];
|
||||||
|
const out = [];
|
||||||
|
|
||||||
|
for(let i = 0; out.length < limit && queues.some(q => q.length); i++) {
|
||||||
|
const queue = queues[i % queues.length];
|
||||||
|
if(queue.length) out.push(queue.shift());
|
||||||
|
}
|
||||||
|
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
function bucketByFirstChar(words) {
|
||||||
|
const buckets = new Map();
|
||||||
|
|
||||||
|
for(const word of words) {
|
||||||
|
const key = word[0];
|
||||||
|
(buckets.get(key) ?? buckets.set(key, []).get(key)).push(word);
|
||||||
|
}
|
||||||
|
|
||||||
|
return buckets;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Finds the closest vocabulary term without comparing obviously unrelated word lengths. */
|
||||||
|
export function bestFuzzyMatch(term, buckets) {
|
||||||
|
const lower = term.toLowerCase();
|
||||||
|
const maxDistance = Math.max(1, Math.ceil(lower.length * 0.34));
|
||||||
|
const candidates = new Set();
|
||||||
|
|
||||||
|
// Check every character so a typo in the first character doesn't eliminate the correct word.
|
||||||
|
for(const char of lower) {
|
||||||
|
const bucket = buckets.get(char);
|
||||||
|
if(bucket) for(const candidate of bucket) candidates.add(candidate);
|
||||||
|
}
|
||||||
|
|
||||||
|
let best = null;
|
||||||
|
let bestScore = 0;
|
||||||
|
|
||||||
|
for(const candidate of candidates) {
|
||||||
|
if(Math.abs(candidate.length - lower.length) > maxDistance) continue;
|
||||||
|
|
||||||
|
const score = fuzzyMatch(candidate, lower).max;
|
||||||
|
|
||||||
|
if(score > bestScore) {
|
||||||
|
bestScore = score;
|
||||||
|
best = candidate;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return best;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Suggests corrected terms from the supplied title vocabulary. */
|
||||||
|
export function suggestCorrection(termList, vocabulary) {
|
||||||
|
if(!vocabulary.size) return null;
|
||||||
|
|
||||||
|
const buckets = bucketByFirstChar(vocabulary);
|
||||||
|
let changed = false;
|
||||||
|
|
||||||
|
const corrected = termList.map(term => {
|
||||||
|
if(vocabulary.has(term.toLowerCase())) return term;
|
||||||
|
|
||||||
|
const fix = bestFuzzyMatch(term, buckets);
|
||||||
|
|
||||||
|
if(fix && fix !== term.toLowerCase()) {
|
||||||
|
changed = true;
|
||||||
|
return fix;
|
||||||
|
}
|
||||||
|
|
||||||
|
return term;
|
||||||
|
});
|
||||||
|
|
||||||
|
return changed ? corrected : null;
|
||||||
|
}
|
||||||
+190
-60
@@ -6,13 +6,16 @@ import net from 'node:net';
|
|||||||
import fs from 'node:fs';
|
import fs from 'node:fs';
|
||||||
import path from 'node:path';
|
import path from 'node:path';
|
||||||
import {fileURLToPath} from 'node:url';
|
import {fileURLToPath} from 'node:url';
|
||||||
import {fromXml} from '@ztimson/utils';
|
import {fromXml, makeArray} from '@ztimson/utils';
|
||||||
|
import {titleVocabulary} from './reader.js';
|
||||||
|
import {diversify, rerank, rrfMerge, suggestCorrection} from './search.js';
|
||||||
|
|
||||||
const execFileAsync = promisify(execFile);
|
const execFileAsync = promisify(execFile);
|
||||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||||
const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin'); // npm package root/bin - where bin/install.js drops the kiwix-tools binaries
|
const DEFAULT_BIN_DIR = path.join(__dirname, '..', 'bin');
|
||||||
const READY_TIMEOUT = 10_000;
|
const READY_TIMEOUT = 10_000;
|
||||||
const READY_POLL_INTERVAL = 100;
|
const READY_POLL_INTERVAL = 100;
|
||||||
|
const WATCH_DEBOUNCE = 300;
|
||||||
|
|
||||||
function findFreePort() {
|
function findFreePort() {
|
||||||
return new Promise((resolve, reject) => {
|
return new Promise((resolve, reject) => {
|
||||||
@@ -25,7 +28,10 @@ function findFreePort() {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Owns a kiwix-serve process's full lifecycle: library.xml, start/stop/reload, content + search access. */
|
function tokenize(terms) {
|
||||||
|
return String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
|
||||||
|
}
|
||||||
|
|
||||||
export class KiwixServer {
|
export class KiwixServer {
|
||||||
static #empty = '<?xml version="1.0" encoding="UTF-8" ?>\n<library version="20110515"></library>\n';
|
static #empty = '<?xml version="1.0" encoding="UTF-8" ?>\n<library version="20110515"></library>\n';
|
||||||
|
|
||||||
@@ -35,15 +41,14 @@ export class KiwixServer {
|
|||||||
#binDir;
|
#binDir;
|
||||||
#libraryPath;
|
#libraryPath;
|
||||||
#child = null;
|
#child = null;
|
||||||
#remote; // baseUrl string if attached to an externally-managed kiwix-serve, else null
|
#remote;
|
||||||
|
#watcher = null;
|
||||||
|
#watchTimer = null;
|
||||||
|
|
||||||
get port() { return this.#port; }
|
get port() { return this.#port; }
|
||||||
get running() { return !!this.#remote || !!this.#child; }
|
get running() { return !!this.#remote || !!this.#child; }
|
||||||
get baseUrl() { return this.#remote || (this.#child ? `http://${this.#host}:${this.#port}` : null); }
|
get baseUrl() { return this.#remote || (this.#child ? `http://${this.#host}:${this.#port}` : null); }
|
||||||
|
|
||||||
/** @param {{port?: number, host?: string, binDir?: string, url?: string}} [opts]
|
|
||||||
* url: attach to an already-running kiwix-serve (e.g. one started elsewhere in your codebase) instead of
|
|
||||||
* spawning/owning one - start/stop/reload become no-ops, and library.xml is read over HTTP instead of disk. */
|
|
||||||
constructor(dir, {port, host = '127.0.0.1', binDir = DEFAULT_BIN_DIR, url} = {}) {
|
constructor(dir, {port, host = '127.0.0.1', binDir = DEFAULT_BIN_DIR, url} = {}) {
|
||||||
this.#dir = dir;
|
this.#dir = dir;
|
||||||
this.#host = host;
|
this.#host = host;
|
||||||
@@ -51,25 +56,29 @@ export class KiwixServer {
|
|||||||
this.#binDir = binDir;
|
this.#binDir = binDir;
|
||||||
this.#libraryPath = path.join(dir, 'library.xml');
|
this.#libraryPath = path.join(dir, 'library.xml');
|
||||||
this.#remote = url ? url.replace(/\/$/, '') : null;
|
this.#remote = url ? url.replace(/\/$/, '') : null;
|
||||||
if (!this.#remote) this.#ensureLocalStore();
|
|
||||||
|
if(!this.#remote) this.#ensureLocalStore();
|
||||||
}
|
}
|
||||||
|
|
||||||
#ensureLocalStore() {
|
#ensureLocalStore() {
|
||||||
fs.mkdirSync(this.#dir, {recursive: true});
|
fs.mkdirSync(this.#dir, {recursive: true});
|
||||||
if (!fs.existsSync(this.#libraryPath)) fs.writeFileSync(this.#libraryPath, KiwixServer.#empty);
|
|
||||||
|
if(!fs.existsSync(this.#libraryPath))
|
||||||
|
fs.writeFileSync(this.#libraryPath, KiwixServer.#empty);
|
||||||
}
|
}
|
||||||
|
|
||||||
#assertRunning() {
|
#assertRunning() {
|
||||||
if (!this.running) throw new Error('KiwixServer is not running - call start() first');
|
if(!this.running) throw new Error('KiwixServer is not running - call start() first');
|
||||||
}
|
}
|
||||||
|
|
||||||
#bin(name) {
|
#bin(name) {
|
||||||
return path.join(this.#binDir, process.platform === 'win32' ? `${name}.exe` : name);
|
return path.join(this.#binDir, process.platform === 'win32' ? `${name}.exe` : name);
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Reads library.xml from disk if we own the server, or over HTTP if attached to a remote one. */
|
|
||||||
async #fetchLibraryXml() {
|
async #fetchLibraryXml() {
|
||||||
if (!this.#remote) return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => '');
|
if(!this.#remote)
|
||||||
|
return fs.promises.readFile(this.#libraryPath, 'utf8').catch(() => '');
|
||||||
|
|
||||||
try {
|
try {
|
||||||
const res = await fetch(`${this.#remote}/library.xml`);
|
const res = await fetch(`${this.#remote}/library.xml`);
|
||||||
return res.ok ? await res.text() : '';
|
return res.ok ? await res.text() : '';
|
||||||
@@ -78,57 +87,108 @@ export class KiwixServer {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Rebuilds library.xml from scratch by scanning `dir` for .zim files - no-op if attached to a remote server. */
|
|
||||||
async #rebuildLibrary() {
|
async #rebuildLibrary() {
|
||||||
if (this.#remote) return;
|
if(this.#remote) return;
|
||||||
|
|
||||||
await fs.promises.rm(this.#libraryPath, {force: true});
|
await fs.promises.rm(this.#libraryPath, {force: true});
|
||||||
|
|
||||||
const files = await this.#zimFiles();
|
const files = await this.#zimFiles();
|
||||||
if (!files.length) return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty);
|
|
||||||
for (const f of files) await execFileAsync(this.#bin('kiwix-manage'), [this.#libraryPath, 'add', path.join(this.#dir, f)]);
|
if(!files.length)
|
||||||
|
return fs.promises.writeFile(this.#libraryPath, KiwixServer.#empty);
|
||||||
|
|
||||||
|
for(const file of files)
|
||||||
|
await execFileAsync(this.#bin('kiwix-manage'), [
|
||||||
|
this.#libraryPath,
|
||||||
|
'add',
|
||||||
|
path.join(this.#dir, file),
|
||||||
|
]);
|
||||||
}
|
}
|
||||||
|
|
||||||
async #waitUntilReady() {
|
async #waitUntilReady() {
|
||||||
const deadline = Date.now() + READY_TIMEOUT;
|
const deadline = Date.now() + READY_TIMEOUT;
|
||||||
while (Date.now() < deadline) {
|
|
||||||
|
while(Date.now() < deadline) {
|
||||||
try {
|
try {
|
||||||
await fetch(`http://${this.#host}:${this.#port}/`);
|
await fetch(`http://${this.#host}:${this.#port}/`);
|
||||||
return;
|
return;
|
||||||
} catch {}
|
} catch {}
|
||||||
|
|
||||||
await new Promise(r => setTimeout(r, READY_POLL_INTERVAL));
|
await new Promise(r => setTimeout(r, READY_POLL_INTERVAL));
|
||||||
}
|
}
|
||||||
|
|
||||||
throw new Error('kiwix-serve did not become ready in time');
|
throw new Error('kiwix-serve did not become ready in time');
|
||||||
}
|
}
|
||||||
|
|
||||||
async #zimFiles() {
|
async #zimFiles() {
|
||||||
return (await fs.promises.readdir(this.#dir).catch(() => [])).filter(f => f.endsWith('.zim'));
|
return (await fs.promises.readdir(this.#dir).catch(() => []))
|
||||||
|
.filter(f => f.endsWith('.zim'));
|
||||||
|
}
|
||||||
|
|
||||||
|
#watchDir() {
|
||||||
|
this.#watcher?.close();
|
||||||
|
|
||||||
|
this.#watcher = fs.watch(this.#dir, (_event, filename) => {
|
||||||
|
if(!filename?.endsWith('.zim')) return;
|
||||||
|
|
||||||
|
clearTimeout(this.#watchTimer);
|
||||||
|
this.#watchTimer = setTimeout(() => this.#rebuildLibrary().catch(() => {}), WATCH_DEBOUNCE);
|
||||||
|
});
|
||||||
|
|
||||||
|
this.#watcher.on('error', () => {});
|
||||||
|
}
|
||||||
|
|
||||||
|
#unwatchDir() {
|
||||||
|
clearTimeout(this.#watchTimer);
|
||||||
|
this.#watchTimer = null;
|
||||||
|
this.#watcher?.close();
|
||||||
|
this.#watcher = null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Rebuilds library.xml and starts kiwix-serve. Resolves once the server is responding. */
|
|
||||||
async start() {
|
async start() {
|
||||||
if (this.#remote || this.#child) return;
|
if(this.#remote || this.#child) return;
|
||||||
|
|
||||||
await fs.promises.mkdir(this.#dir, {recursive: true});
|
await fs.promises.mkdir(this.#dir, {recursive: true});
|
||||||
await this.#rebuildLibrary();
|
await this.#rebuildLibrary();
|
||||||
this.#port ??= await findFreePort();
|
this.#port ??= await findFreePort();
|
||||||
|
|
||||||
this.#child = spawn(this.#bin('kiwix-serve'), ['--library', '-i', this.#host, '-p', String(this.#port), this.#libraryPath], {stdio: 'ignore'});
|
this.#child = spawn(this.#bin('kiwix-serve'), [
|
||||||
this.#child.on('exit', () => { this.#child = null; });
|
'--library',
|
||||||
|
'--monitorLibrary',
|
||||||
|
'-i',
|
||||||
|
this.#host,
|
||||||
|
'-p',
|
||||||
|
String(this.#port),
|
||||||
|
this.#libraryPath,
|
||||||
|
], {stdio: 'ignore'});
|
||||||
|
|
||||||
|
this.#child.on('exit', () => {
|
||||||
|
this.#child = null;
|
||||||
|
this.#unwatchDir();
|
||||||
|
});
|
||||||
|
|
||||||
try {
|
try {
|
||||||
await this.#waitUntilReady();
|
await this.#waitUntilReady();
|
||||||
} catch (e) {
|
} catch(e) {
|
||||||
await this.stop();
|
await this.stop();
|
||||||
throw e;
|
throw e;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
this.#watchDir();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Gracefully stops kiwix-serve, if we own it. No-op if attached to a remote instance. */
|
|
||||||
async stop() {
|
async stop() {
|
||||||
if (this.#remote || !this.#child) return;
|
this.#unwatchDir();
|
||||||
|
|
||||||
|
if(this.#remote || !this.#child) return;
|
||||||
|
|
||||||
const child = this.#child;
|
const child = this.#child;
|
||||||
|
|
||||||
await new Promise(resolve => {
|
await new Promise(resolve => {
|
||||||
child.once('exit', resolve);
|
child.once('exit', resolve);
|
||||||
child.kill('SIGTERM');
|
child.kill('SIGTERM');
|
||||||
});
|
});
|
||||||
|
|
||||||
this.#child = null;
|
this.#child = null;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -137,22 +197,23 @@ export class KiwixServer {
|
|||||||
await this.start();
|
await this.start();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Rebuilds library.xml from disk and restarts kiwix-serve. No-op if attached to a remote instance -
|
|
||||||
* whoever owns that process is responsible for reloading it. */
|
|
||||||
async reload() {
|
async reload() {
|
||||||
if (this.#remote || !this.#child) return;
|
if(this.#remote || !this.#child) return;
|
||||||
await this.restart();
|
await this.#rebuildLibrary();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Local catalog listing - same flat shape as the online catalog (catalog.js), plus a `file` field. */
|
|
||||||
async list() {
|
async list() {
|
||||||
this.#assertRunning();
|
this.#assertRunning();
|
||||||
|
|
||||||
const xml = await this.#fetchLibraryXml();
|
const xml = await this.#fetchLibraryXml();
|
||||||
if (!xml) return [];
|
if(!xml) return [];
|
||||||
|
|
||||||
const entries = fromXml(xml);
|
const entries = fromXml(xml);
|
||||||
return (entries?.library?.book || []).map(e => {
|
|
||||||
|
return makeArray(entries?.library?.book || []).map(e => {
|
||||||
const tags = e.tags.split(';');
|
const tags = e.tags.split(';');
|
||||||
const name = e.path.replaceAll('.zim', '');
|
const name = e.path.replace(/\.zim$/i, '');
|
||||||
|
|
||||||
return {
|
return {
|
||||||
id: e.id,
|
id: e.id,
|
||||||
title: e.title,
|
title: e.title,
|
||||||
@@ -167,6 +228,7 @@ export class KiwixServer {
|
|||||||
publisher: e.publisher,
|
publisher: e.publisher,
|
||||||
articleCount: +e.articleCount || 0,
|
articleCount: +e.articleCount || 0,
|
||||||
sizeMb: +(Number(e.size) / 1024).toFixed(1) || 0,
|
sizeMb: +(Number(e.size) / 1024).toFixed(1) || 0,
|
||||||
|
file: e.path,
|
||||||
href: name,
|
href: name,
|
||||||
icon: `data:${e.faviconMimetype || 'image/png'};base64,${e.favicon}`,
|
icon: `data:${e.faviconMimetype || 'image/png'};base64,${e.favicon}`,
|
||||||
viewer: `${this.baseUrl}/content/${name}`,
|
viewer: `${this.baseUrl}/content/${name}`,
|
||||||
@@ -174,58 +236,126 @@ export class KiwixServer {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Splits a href ("zim/path/to/page") or a full content/viewer URL into {zim, path}. */
|
|
||||||
#splitHref(href) {
|
#splitHref(href) {
|
||||||
const clean = href.replace(`${this.baseUrl}/content/`, '').replace(/^\/+/, '');
|
const clean = href.replace(`${this.baseUrl}/content/`, '').replace(/^\/+/, '');
|
||||||
const [zim, ...rest] = clean.split('/');
|
const [zim, ...rest] = clean.split('/');
|
||||||
return {zim, path: rest.join('/')};
|
return {zim, path: rest.join('/')};
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Builds a kiwix-serve content URL from a href (as returned by list()/search()), or from an already-built content/viewer URL. */
|
|
||||||
link(href) {
|
link(href) {
|
||||||
this.#assertRunning();
|
this.#assertRunning();
|
||||||
|
|
||||||
const {zim, path} = this.#splitHref(href);
|
const {zim, path} = this.#splitHref(href);
|
||||||
|
|
||||||
return `${this.baseUrl}/content/${zim}${path ? '/' + path : ''}`;
|
return `${this.baseUrl}/content/${zim}${path ? '/' + path : ''}`;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Fetches a single asset's raw bytes straight from kiwix-serve. */
|
|
||||||
async raw(href) {
|
async raw(href) {
|
||||||
const res = await fetch(this.link(href));
|
const res = await fetch(this.link(href));
|
||||||
|
|
||||||
if(!res.ok) return null;
|
if(!res.ok) return null;
|
||||||
return {mimetype: res.headers.get('content-type'), data: Buffer.from(await res.arrayBuffer())};
|
|
||||||
|
return {
|
||||||
|
mimetype: res.headers.get('content-type'),
|
||||||
|
data: Buffer.from(await res.arrayBuffer()),
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Fulltext search across every local ZIM via kiwix-serve's own xapian index */
|
async #rawSearch(termList, scoped, bookMap, limit) {
|
||||||
async search(terms, limit = 20) {
|
const perBook = await Promise.all(scoped.map(async book => {
|
||||||
this.#assertRunning();
|
const params = new URLSearchParams({
|
||||||
const termList = String(terms).split(/[,\s]+/).map(t => t.trim()).filter(Boolean);
|
pattern: termList.join(' '),
|
||||||
if (!termList.length) return [];
|
format: 'xml',
|
||||||
|
pageLength: String(limit),
|
||||||
|
'books.name': book.href,
|
||||||
|
});
|
||||||
|
|
||||||
const params = new URLSearchParams({pattern: termList.join(' '), format: 'xml', pageLength: String(limit)});
|
|
||||||
const res = await fetch(`${this.baseUrl}/search?${params}`);
|
const res = await fetch(`${this.baseUrl}/search?${params}`);
|
||||||
if (!res.ok) return [];
|
if(!res.ok) return [];
|
||||||
|
|
||||||
const found = fromXml(await res.text())?.rss?.channel?.item || [];
|
const found = fromXml(await res.text())?.rss?.channel?.item || [];
|
||||||
|
|
||||||
|
return found.map(hit => {
|
||||||
|
const resultBook = bookMap.get(hit.book?.title) || book;
|
||||||
|
|
||||||
|
const prefix = `/content/${resultBook.href}/`;
|
||||||
|
const page = hit.link.startsWith(prefix)
|
||||||
|
? hit.link.slice(prefix.length)
|
||||||
|
: hit.link.replace(/^\/+/, '');
|
||||||
|
|
||||||
|
return {
|
||||||
|
id: resultBook.id,
|
||||||
|
title: hit.title,
|
||||||
|
page,
|
||||||
|
name: resultBook.name,
|
||||||
|
publisher: resultBook.publisher,
|
||||||
|
href: `${resultBook.href}/${page}`,
|
||||||
|
icon: resultBook.icon,
|
||||||
|
viewer: this.baseUrl + hit.link,
|
||||||
|
summary: hit.description,
|
||||||
|
xapianScore: +hit.score || 0,
|
||||||
|
};
|
||||||
|
}).filter(Boolean);
|
||||||
|
}));
|
||||||
|
|
||||||
|
return rrfMerge(perBook);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Fulltext search across local ZIMs with ranking, diversification and spelling correction. */
|
||||||
|
async search(terms, {limit = 20, sources = null} = {}) {
|
||||||
|
this.#assertRunning();
|
||||||
|
|
||||||
|
const termList = tokenize(terms);
|
||||||
|
if(!termList.length) return {results: [], spellcheck: null};
|
||||||
|
|
||||||
const books = await this.list();
|
const books = await this.list();
|
||||||
const bookMap = new Map(books.map(b => [b.title, b]));
|
const bookMap = new Map(books.map(b => [b.title, b]));
|
||||||
|
|
||||||
return found.map(hit => {
|
const scoped = sources?.length
|
||||||
const book = bookMap.get(hit.book.title);
|
? books.filter(b => sources.includes(b.name) || sources.includes(b.href))
|
||||||
if (!book) return null;
|
: books;
|
||||||
const prefix = `/content/${book.href}/`;
|
|
||||||
const page = hit.link.startsWith(prefix) ? hit.link.slice(prefix.length) : hit.link.replace(/^\/+/, '');
|
if(!scoped.length) return {results: [], spellcheck: null};
|
||||||
|
|
||||||
|
let activeTerms = termList;
|
||||||
|
let hits = await this.#rawSearch(activeTerms, scoped, bookMap, limit);
|
||||||
|
let spellcheck = null;
|
||||||
|
|
||||||
|
if(!hits.length && !this.#remote) {
|
||||||
|
const vocabulary = new Set();
|
||||||
|
|
||||||
|
for(const book of scoped) {
|
||||||
|
if(!book.file) continue;
|
||||||
|
|
||||||
|
try {
|
||||||
|
for(const word of await titleVocabulary(path.join(this.#dir, book.file)))
|
||||||
|
vocabulary.add(word);
|
||||||
|
} catch {
|
||||||
|
// Unreadable ZIM - skip it, don't fail the whole search.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const corrected = suggestCorrection(termList, vocabulary);
|
||||||
|
|
||||||
|
if(corrected) {
|
||||||
|
const retry = await this.#rawSearch(corrected, scoped, bookMap, limit);
|
||||||
|
|
||||||
|
if(retry.length) {
|
||||||
|
hits = retry;
|
||||||
|
activeTerms = corrected;
|
||||||
|
spellcheck = {
|
||||||
|
from: termList.join(' '),
|
||||||
|
to: corrected.join(' '),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const ranked = rerank(hits, activeTerms);
|
||||||
|
|
||||||
return {
|
return {
|
||||||
id: book.id,
|
results: diversify(ranked, limit),
|
||||||
title: hit.title,
|
spellcheck,
|
||||||
page,
|
|
||||||
name: book.name,
|
|
||||||
publisher: book.publisher,
|
|
||||||
href: `${book.href}/${page}`,
|
|
||||||
icon: book.icon,
|
|
||||||
viewer: this.baseUrl + hit.link,
|
|
||||||
summary: hit.description,
|
|
||||||
score: +hit.score || 0,
|
|
||||||
};
|
};
|
||||||
}).filter(Boolean);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+12
-15
@@ -1,39 +1,36 @@
|
|||||||
export function levenshtein(a, b) {
|
export function levenshtein(a, b) {
|
||||||
const m = a.length, n = b.length;
|
const m = a.length, n = b.length;
|
||||||
if (!m) return n;
|
if(!m) return n;
|
||||||
if (!n) return m;
|
if(!n) return m;
|
||||||
const dp = Array.from({length: m + 1}, (_, i) => [i, ...Array(n).fill(0)]);
|
const dp = Array.from({length: m + 1}, (_, i) => [i, ...Array(n).fill(0)]);
|
||||||
for (let j = 0; j <= n; j++) dp[0][j] = j;
|
for(let j = 0; j <= n; j++) dp[0][j] = j;
|
||||||
for (let i = 1; i <= m; i++) {
|
for(let i = 1; i <= m; i++) {
|
||||||
for (let j = 1; j <= n; j++) {
|
for(let j = 1; j <= n; j++) {
|
||||||
dp[i][j] = a[i - 1] === b[j - 1]
|
dp[i][j] = a[i - 1] === b[j - 1] ? dp[i - 1][j - 1] : 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
|
||||||
? dp[i - 1][j - 1]
|
|
||||||
: 1 + Math.min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1]);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return dp[m][n];
|
return dp[m][n];
|
||||||
}
|
}
|
||||||
|
|
||||||
function scoreAgainst(text, term) {
|
function scoreAgainst(text, term) {
|
||||||
if (text.includes(term)) return 1 - (text.length - term.length) / text.length * 0.3;
|
if (!text.length || !term.length) return 0;
|
||||||
if (!text.length || !term.length || text[0] !== term[0]) return 0;
|
if (text === term) return 1;
|
||||||
const dist = levenshtein(text, term);
|
if (text.includes(term)) return 0.8;
|
||||||
const maxAllowed = Math.max(1, Math.ceil(term.length * 0.34));
|
const maxAllowed = Math.max(1, Math.ceil(term.length * 0.34));
|
||||||
|
if (Math.abs(text.length - term.length) > maxAllowed) return 0;
|
||||||
|
const dist = levenshtein(text, term);
|
||||||
if (dist > maxAllowed) return 0;
|
if (dist > maxAllowed) return 0;
|
||||||
return 1 - dist / Math.max(text.length, term.length);
|
return 1 - dist / Math.max(text.length, term.length);
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Compares `target` against one or more search terms; returns avg/max/per-term similarity. */
|
|
||||||
export function fuzzyMatch(target, ...terms) {
|
export function fuzzyMatch(target, ...terms) {
|
||||||
if (!terms.length) throw new Error('Requires at least 1 term to compare');
|
if (!terms.length) throw new Error('Requires at least 1 term to compare');
|
||||||
const lowerTarget = String(target).toLowerCase();
|
const lowerTarget = String(target).toLowerCase();
|
||||||
const words = lowerTarget.split(/\W+/).filter(Boolean);
|
const words = lowerTarget.split(/\W+/).filter(Boolean);
|
||||||
|
|
||||||
const similarities = terms.map(term => {
|
const similarities = terms.map(term => {
|
||||||
const t = term.toLowerCase();
|
const t = String(term).toLowerCase();
|
||||||
return Math.max(scoreAgainst(lowerTarget, t), ...words.map(w => scoreAgainst(w, t)));
|
return Math.max(scoreAgainst(lowerTarget, t), ...words.map(w => scoreAgainst(w, t)));
|
||||||
});
|
});
|
||||||
|
|
||||||
return {
|
return {
|
||||||
avg: similarities.reduce((acc, s) => acc + s, 0) / similarities.length,
|
avg: similarities.reduce((acc, s) => acc + s, 0) / similarities.length,
|
||||||
max: Math.max(...similarities),
|
max: Math.max(...similarities),
|
||||||
|
|||||||
Reference in New Issue
Block a user