Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 73d6ee0f2a | |||
| bee4085666 | |||
| 3b5c71de7c | |||
| 8229e02a52 | |||
| a6fb8ae828 |
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@ztimson/ai-utils",
|
"name": "@ztimson/ai-utils",
|
||||||
"version": "1.2.1",
|
"version": "1.2.5",
|
||||||
"description": "AI Utility library",
|
"description": "AI Utility library",
|
||||||
"author": "Zak Timson",
|
"author": "Zak Timson",
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
|
|||||||
334
src/kd-tree.ts
Normal file
334
src/kd-tree.ts
Normal file
@@ -0,0 +1,334 @@
|
|||||||
|
export type DistanceMetric = "euclidean" | "cosine";
|
||||||
|
|
||||||
|
export interface KDPoint<T = unknown> {
|
||||||
|
vector: number[];
|
||||||
|
payload: T;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface KNNResult<T = unknown> {
|
||||||
|
point: KDPoint<T>;
|
||||||
|
distance: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
interface KDNode<T> {
|
||||||
|
point: KDPoint<T>;
|
||||||
|
axis: number;
|
||||||
|
left: KDNode<T> | null;
|
||||||
|
right: KDNode<T> | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ─── Distance helpers ─────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
function euclidean(a: number[], b: number[]): number {
|
||||||
|
let sum = 0;
|
||||||
|
for (let i = 0; i < a.length; i++) {
|
||||||
|
const d = a[i] - b[i];
|
||||||
|
sum += d * d;
|
||||||
|
}
|
||||||
|
return Math.sqrt(sum);
|
||||||
|
}
|
||||||
|
|
||||||
|
function cosine(a: number[], b: number[]): number {
|
||||||
|
let dot = 0, normA = 0, normB = 0;
|
||||||
|
for (let i = 0; i < a.length; i++) {
|
||||||
|
dot += a[i] * b[i];
|
||||||
|
normA += a[i] * a[i];
|
||||||
|
normB += b[i] * b[i];
|
||||||
|
}
|
||||||
|
const denom = Math.sqrt(normA) * Math.sqrt(normB);
|
||||||
|
return denom === 0 ? 1 : 1 - dot / denom; // distance = 1 - similarity
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Keeps the k closest candidates in memory, evicts the furthest when full
|
||||||
|
*/
|
||||||
|
class BoundedMaxHeap<T> {
|
||||||
|
private heap: KNNResult<T>[] = [];
|
||||||
|
|
||||||
|
constructor(private readonly k: number) {}
|
||||||
|
|
||||||
|
get size(): number { return this.heap.length; }
|
||||||
|
|
||||||
|
get worstDistance(): number {
|
||||||
|
return this.heap.length < this.k ? Infinity : this.heap[0].distance;
|
||||||
|
}
|
||||||
|
|
||||||
|
push(item: KNNResult<T>): void {
|
||||||
|
if (this.heap.length < this.k) {
|
||||||
|
this.heap.push(item);
|
||||||
|
this.bubbleUp(this.heap.length - 1);
|
||||||
|
} else if (item.distance < this.heap[0].distance) {
|
||||||
|
this.heap[0] = item;
|
||||||
|
this.sinkDown(0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
toSortedArray(): KNNResult<T>[] {
|
||||||
|
return [...this.heap].sort((a, b) => a.distance - b.distance);
|
||||||
|
}
|
||||||
|
|
||||||
|
private bubbleUp(i: number): void {
|
||||||
|
while (i > 0) {
|
||||||
|
const parent = (i - 1) >> 1;
|
||||||
|
if (this.heap[parent].distance >= this.heap[i].distance) break;
|
||||||
|
[this.heap[parent], this.heap[i]] = [this.heap[i], this.heap[parent]];
|
||||||
|
i = parent;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private sinkDown(i: number): void {
|
||||||
|
const n = this.heap.length;
|
||||||
|
while (true) {
|
||||||
|
let largest = i;
|
||||||
|
const l = 2 * i + 1, r = 2 * i + 2;
|
||||||
|
if (l < n && this.heap[l].distance > this.heap[largest].distance) largest = l;
|
||||||
|
if (r < n && this.heap[r].distance > this.heap[largest].distance) largest = r;
|
||||||
|
if (largest === i) break;
|
||||||
|
[this.heap[largest], this.heap[i]] = [this.heap[i], this.heap[largest]];
|
||||||
|
i = largest;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* K-D Tree for efficient nearest-neighbor search over high-dimensional vectors / embeddings.
|
||||||
|
*
|
||||||
|
* Supports:
|
||||||
|
* - Insertion of labeled points
|
||||||
|
* - k-nearest-neighbor (KNN) search
|
||||||
|
* - Radius search (all points within a given distance)
|
||||||
|
* - Euclidean and cosine distance metrics
|
||||||
|
* - Bulk construction (balanced tree) for best query performance
|
||||||
|
*/
|
||||||
|
export class KDTree<T = unknown> {
|
||||||
|
private root: KDNode<T> | null = null;
|
||||||
|
private _size = 0;
|
||||||
|
private readonly dims: number;
|
||||||
|
private readonly distanceFn: (a: number[], b: number[]) => number;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @param dims Dimensionality of all vectors (must be consistent).
|
||||||
|
* @param metric Distance metric to use. Default: "euclidean".
|
||||||
|
* @param points Optional initial set of points. Builds a balanced tree
|
||||||
|
* in O(n log² n) — prefer this over inserting one-by-one
|
||||||
|
* when you have a large corpus.
|
||||||
|
*/
|
||||||
|
constructor(
|
||||||
|
dims: number,
|
||||||
|
metric: DistanceMetric = "euclidean",
|
||||||
|
points?: KDPoint<T>[]
|
||||||
|
) {
|
||||||
|
this.dims = dims;
|
||||||
|
this.distanceFn = metric === "cosine" ? cosine : euclidean;
|
||||||
|
|
||||||
|
if (points && points.length > 0) {
|
||||||
|
this.validateAll(points);
|
||||||
|
this.root = this.buildBalanced([...points], 0);
|
||||||
|
this._size = points.length;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Total number of points stored in the tree. */
|
||||||
|
get size(): number { return this._size; }
|
||||||
|
|
||||||
|
// ── Insertion ──────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Insert a single point. O(log n) average, O(n) worst case on skewed data.
|
||||||
|
* For bulk loading prefer passing points to the constructor.
|
||||||
|
*/
|
||||||
|
insert(point: KDPoint<T>): void {
|
||||||
|
this.validate(point);
|
||||||
|
this.root = this.insertNode(this.root, point, 0);
|
||||||
|
this._size++;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── KNN search ─────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Find the k nearest neighbors to `query`.
|
||||||
|
* Returns results sorted by distance ascending.
|
||||||
|
*/
|
||||||
|
knn(query: number[], k: number): KNNResult<T>[] {
|
||||||
|
if (k <= 0) throw new RangeError("k must be a positive integer");
|
||||||
|
this.validateVector(query);
|
||||||
|
|
||||||
|
const heap = new BoundedMaxHeap<T>(k);
|
||||||
|
this.searchKNN(this.root, query, k, heap, 0);
|
||||||
|
return heap.toSortedArray();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Nearest single neighbor. Convenience wrapper around knn(query, 1).
|
||||||
|
* Returns null if the tree is empty.
|
||||||
|
*/
|
||||||
|
nearest(query: number[]): KNNResult<T> | null {
|
||||||
|
const results = this.knn(query, 1);
|
||||||
|
return results[0] ?? null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Radius search ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Return all points whose distance to `query` is ≤ `radius`,
|
||||||
|
* sorted by distance ascending.
|
||||||
|
*/
|
||||||
|
radiusSearch(query: number[], radius: number): KNNResult<T>[] {
|
||||||
|
if (radius < 0) throw new RangeError("radius must be non-negative");
|
||||||
|
this.validateVector(query);
|
||||||
|
|
||||||
|
const results: KNNResult<T>[] = [];
|
||||||
|
this.searchRadius(this.root, query, radius, results, 0);
|
||||||
|
results.sort((a, b) => a.distance - b.distance);
|
||||||
|
return results;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Conversion ─────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/** Collect all points in the tree (order not guaranteed). */
|
||||||
|
toArray(): KDPoint<T>[] {
|
||||||
|
const out: KDPoint<T>[] = [];
|
||||||
|
this.collect(this.root, out);
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Rebuild the tree from its current points as a balanced tree.
|
||||||
|
* Useful after many individual insertions to restore O(log n) query time.
|
||||||
|
*/
|
||||||
|
rebalance(): void {
|
||||||
|
const points = this.toArray();
|
||||||
|
this.root = points.length ? this.buildBalanced(points, 0) : null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Private: build ─────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
private buildBalanced(points: KDPoint<T>[], depth: number): KDNode<T> {
|
||||||
|
const axis = depth % this.dims;
|
||||||
|
points.sort((a, b) => a.vector[axis] - b.vector[axis]);
|
||||||
|
|
||||||
|
const mid = Math.floor(points.length / 2);
|
||||||
|
return {
|
||||||
|
point: points[mid],
|
||||||
|
axis,
|
||||||
|
left: points.slice(0, mid).length
|
||||||
|
? this.buildBalanced(points.slice(0, mid), depth + 1)
|
||||||
|
: null,
|
||||||
|
right: points.slice(mid + 1).length
|
||||||
|
? this.buildBalanced(points.slice(mid + 1), depth + 1)
|
||||||
|
: null,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Private: insert ────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
private insertNode(
|
||||||
|
node: KDNode<T> | null,
|
||||||
|
point: KDPoint<T>,
|
||||||
|
depth: number
|
||||||
|
): KDNode<T> {
|
||||||
|
if (node === null) {
|
||||||
|
return { point, axis: depth % this.dims, left: null, right: null };
|
||||||
|
}
|
||||||
|
const axis = depth % this.dims;
|
||||||
|
if (point.vector[axis] < node.point.vector[axis]) {
|
||||||
|
node.left = this.insertNode(node.left, point, depth + 1);
|
||||||
|
} else {
|
||||||
|
node.right = this.insertNode(node.right, point, depth + 1);
|
||||||
|
}
|
||||||
|
return node;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Private: KNN traversal ─────────────────────────────────────────────────
|
||||||
|
|
||||||
|
private searchKNN(
|
||||||
|
node: KDNode<T> | null,
|
||||||
|
query: number[],
|
||||||
|
k: number,
|
||||||
|
heap: BoundedMaxHeap<T>,
|
||||||
|
depth: number
|
||||||
|
): void {
|
||||||
|
if (node === null) return;
|
||||||
|
|
||||||
|
const dist = this.distanceFn(query, node.point.vector);
|
||||||
|
heap.push({ point: node.point, distance: dist });
|
||||||
|
|
||||||
|
const axis = node.axis;
|
||||||
|
const diff = query[axis] - node.point.vector[axis];
|
||||||
|
const [near, far] = diff <= 0
|
||||||
|
? [node.left, node.right]
|
||||||
|
: [node.right, node.left];
|
||||||
|
|
||||||
|
this.searchKNN(near, query, k, heap, depth + 1);
|
||||||
|
|
||||||
|
// Only explore the far side if it could contain a closer point.
|
||||||
|
// For cosine distance we can't prune by axis gap alone, so always explore.
|
||||||
|
const shouldExplore =
|
||||||
|
this.distanceFn === cosine
|
||||||
|
? true
|
||||||
|
: Math.abs(diff) < heap.worstDistance;
|
||||||
|
|
||||||
|
if (shouldExplore) {
|
||||||
|
this.searchKNN(far, query, k, heap, depth + 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Private: radius traversal ──────────────────────────────────────────────
|
||||||
|
|
||||||
|
private searchRadius(
|
||||||
|
node: KDNode<T> | null,
|
||||||
|
query: number[],
|
||||||
|
radius: number,
|
||||||
|
results: KNNResult<T>[],
|
||||||
|
depth: number
|
||||||
|
): void {
|
||||||
|
if (node === null) return;
|
||||||
|
|
||||||
|
const dist = this.distanceFn(query, node.point.vector);
|
||||||
|
if (dist <= radius) {
|
||||||
|
results.push({ point: node.point, distance: dist });
|
||||||
|
}
|
||||||
|
|
||||||
|
const axis = node.axis;
|
||||||
|
const diff = query[axis] - node.point.vector[axis];
|
||||||
|
const [near, far] = diff <= 0
|
||||||
|
? [node.left, node.right]
|
||||||
|
: [node.right, node.left];
|
||||||
|
|
||||||
|
this.searchRadius(near, query, radius, results, depth + 1);
|
||||||
|
|
||||||
|
const shouldExplore =
|
||||||
|
this.distanceFn === cosine ? true : Math.abs(diff) <= radius;
|
||||||
|
|
||||||
|
if (shouldExplore) {
|
||||||
|
this.searchRadius(far, query, radius, results, depth + 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Private: collect ───────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
private collect(node: KDNode<T> | null, out: KDPoint<T>[]): void {
|
||||||
|
if (node === null) return;
|
||||||
|
out.push(node.point);
|
||||||
|
this.collect(node.left, out);
|
||||||
|
this.collect(node.right, out);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Private: validation ────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
private validateVector(v: number[]): void {
|
||||||
|
if (v.length !== this.dims) {
|
||||||
|
throw new TypeError(
|
||||||
|
`Vector length ${v.length} does not match tree dimensionality ${this.dims}`
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private validate(point: KDPoint<T>): void {
|
||||||
|
this.validateVector(point.vector);
|
||||||
|
}
|
||||||
|
|
||||||
|
private validateAll(points: KDPoint<T>[]): void {
|
||||||
|
for (const p of points) this.validate(p);
|
||||||
|
}
|
||||||
|
}
|
||||||
29
src/llm.ts
29
src/llm.ts
@@ -6,7 +6,7 @@ import {AiTool, AiToolArg} from './tools.ts';
|
|||||||
import {fileURLToPath} from 'url';
|
import {fileURLToPath} from 'url';
|
||||||
import {dirname, join} from 'path';
|
import {dirname, join} from 'path';
|
||||||
import {spawn} from 'node:child_process';
|
import {spawn} from 'node:child_process';
|
||||||
import {Memory, MemoryManager} from './memory.ts';
|
import {Memory, MemoryCache, MemoryManager} from './memory.ts';
|
||||||
|
|
||||||
export type AnthropicConfig = {proto: 'anthropic', token: string};
|
export type AnthropicConfig = {proto: 'anthropic', token: string};
|
||||||
export type OpenAiConfig = {proto: 'openai', host?: string, token: string};
|
export type OpenAiConfig = {proto: 'openai', host?: string, token: string};
|
||||||
@@ -55,7 +55,7 @@ export type LLMRequest = {
|
|||||||
/** Compress old messages in the chat to free up context */
|
/** Compress old messages in the chat to free up context */
|
||||||
compress?: {max: number; min: number};
|
compress?: {max: number; min: number};
|
||||||
/** User's memory documents - RAG injected automatically each turn */
|
/** User's memory documents - RAG injected automatically each turn */
|
||||||
memory?: Memory[];
|
memory?: Memory[] | MemoryCache;
|
||||||
/** Model to use for memory operations */
|
/** Model to use for memory operations */
|
||||||
memoryModel?: string;
|
memoryModel?: string;
|
||||||
/** Skill documents the AI can browse and read on demand */
|
/** Skill documents the AI can browse and read on demand */
|
||||||
@@ -191,18 +191,19 @@ class LLM {
|
|||||||
|
|
||||||
// Memory
|
// Memory
|
||||||
if (options.memory) {
|
if (options.memory) {
|
||||||
const relevant = await this.memoryManager.recollect(message, options.memory, 1);
|
const mems = options.memory instanceof MemoryCache ? options.memory.memories : options.memory;
|
||||||
|
const relevant = await this.memoryManager.recollect(message, options.memory, 5);
|
||||||
prompts.unshift(`You have access to the following memory files:
|
prompts.unshift(`You have access to the following memory files:
|
||||||
${options.memory.map(m => `- ${m.name}: ${m.description}`).join('\n')}
|
${mems.map(m => `- ${m.name}: ${m.description}`).join('\n')}
|
||||||
${relevant.length ? `
|
${relevant.length ? `
|
||||||
The closest memory has been added primitively:
|
Relevant memories have been preloaded:
|
||||||
\`\`\`
|
${relevant.map(r => `
|
||||||
Name: ${relevant[0].name}
|
**${r.name}**
|
||||||
Description: ${relevant[0].description}
|
${r.description}
|
||||||
${relevant[0].content}
|
${r.content}
|
||||||
\`\`\`
|
`).join('\n---\n')}
|
||||||
` : ''}`.trim());
|
` : ''}`.trim());
|
||||||
tools.push(this.memoryManager.tools.read(<Memory[]>options.memory));
|
tools.push(this.memoryManager.tools.read(options.memory));
|
||||||
}
|
}
|
||||||
|
|
||||||
prompts.unshift(options.system || this.ai.options.llm?.system || '');
|
prompts.unshift(options.system || this.ai.options.llm?.system || '');
|
||||||
@@ -215,7 +216,7 @@ ${relevant[0].content}
|
|||||||
|
|
||||||
// Auto-memorize before compressing
|
// Auto-memorize before compressing
|
||||||
if(options.compress && this.estimateTokens(history) >= options.compress.max) {
|
if(options.compress && this.estimateTokens(history) >= options.compress.max) {
|
||||||
if(options.memory) await this.memoryManager.memorize(history, options.memory, options);
|
if(options.memory) await this.memoryManager.memorize(history, options.memory, {model: options.memoryModel || this.defaultModel, ...options});
|
||||||
const compressed = await this.compressHistory(history, options.compress.max, options.compress.min, options);
|
const compressed = await this.compressHistory(history, options.compress.max, options.compress.min, options);
|
||||||
if(options.history) options.history.splice(0, options.history.length, ...compressed);
|
if(options.history) options.history.splice(0, options.history.length, ...compressed);
|
||||||
}
|
}
|
||||||
@@ -228,8 +229,8 @@ ${relevant[0].content}
|
|||||||
* Digest full conversation history into memory documents.
|
* Digest full conversation history into memory documents.
|
||||||
* Call on session end to persist the conversation.
|
* Call on session end to persist the conversation.
|
||||||
*/
|
*/
|
||||||
async updateMemory(history: LLMMessage[], memories: Memory[], options: LLMRequest = {}): Promise<void> {
|
async updateMemory(history: LLMMessage[], memories: Memory[] | MemoryCache, options: LLMRequest = {}): Promise<Memory[]> {
|
||||||
await this.memoryManager.memorize(history, memories, {model: this.defaultModel, ...options});
|
return this.memoryManager.memorize(history, memories, {model: this.defaultModel, ...options});
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
705
src/memory.ts
705
src/memory.ts
@@ -1,177 +1,618 @@
|
|||||||
// memory.ts
|
|
||||||
import {LLMRequest, LLMMessage} from './llm.ts';
|
import {LLMRequest, LLMMessage} from './llm.ts';
|
||||||
import {AiTool} from './tools.ts';
|
import {AiTool} from './tools.ts';
|
||||||
|
import {KDTree, KDPoint} from './kd-tree.ts';
|
||||||
|
|
||||||
/** Background information the AI will be fed as a knowledge document */
|
|
||||||
export type Memory = {
|
export type Memory = {
|
||||||
/** Memory subject */
|
|
||||||
name: string;
|
name: string;
|
||||||
/** Short description of what this document contains - used for RAG retrieval */
|
|
||||||
description: string;
|
description: string;
|
||||||
/** Full markdown content of the document */
|
|
||||||
content: string;
|
content: string;
|
||||||
/** Embedding vector of the description - used for similarity search */
|
|
||||||
embedding: number[];
|
embedding: number[];
|
||||||
}
|
}
|
||||||
|
|
||||||
export type MemoryCollection = {
|
type MemoryRef = {
|
||||||
/** Memory subject */
|
|
||||||
name: string;
|
name: string;
|
||||||
/** Short description - required if isNew */
|
description: string;
|
||||||
description?: string;
|
}
|
||||||
/** Extracted facts to merge */
|
|
||||||
|
type FactBucket = {
|
||||||
|
subject: string;
|
||||||
facts: string[];
|
facts: string[];
|
||||||
|
isNew: boolean;
|
||||||
|
}
|
||||||
|
|
||||||
|
export type MemoryNode = {
|
||||||
|
name: string;
|
||||||
|
missing: boolean;
|
||||||
|
links: string[];
|
||||||
|
backlinks: string[];
|
||||||
|
}
|
||||||
|
|
||||||
|
export function buildMemoryGraph(memories: Memory[] | MemoryCache): MemoryNode[] {
|
||||||
|
const mems = memories instanceof MemoryCache ? memories.memories : memories;
|
||||||
|
const nameSet = new Set(mems.map(m => m.name));
|
||||||
|
const ghosts = new Set<string>();
|
||||||
|
|
||||||
|
const nodes: MemoryNode[] = mems.map(m => {
|
||||||
|
const {links, backlinks} = extractMetadata(m.content);
|
||||||
|
return {
|
||||||
|
name: m.name,
|
||||||
|
missing: false,
|
||||||
|
links,
|
||||||
|
backlinks,
|
||||||
|
};
|
||||||
|
});
|
||||||
|
|
||||||
|
for (const node of nodes) {
|
||||||
|
for (const link of node.links) {
|
||||||
|
if (!nameSet.has(link)) ghosts.add(link);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return [
|
||||||
|
...nodes,
|
||||||
|
...[...ghosts].map(name => ({
|
||||||
|
name,
|
||||||
|
missing: true,
|
||||||
|
links: [],
|
||||||
|
backlinks: nodes
|
||||||
|
.filter(n => n.links.includes(name))
|
||||||
|
.map(n => n.name),
|
||||||
|
}))
|
||||||
|
];
|
||||||
|
}
|
||||||
|
|
||||||
|
function extractLinks(content: string): string[] {
|
||||||
|
const matches = content.matchAll(/\[\[([^\]]+)\]\]/g);
|
||||||
|
return [...new Set([...matches].map(m => m[1].trim()))];
|
||||||
|
}
|
||||||
|
|
||||||
|
export function extractMetadata(content: string): {links: string[], backlinks: string[]} {
|
||||||
|
const match = content.match(/^---\n([\s\S]*?)\n---/);
|
||||||
|
if (!match) return {links: [], backlinks: []};
|
||||||
|
|
||||||
|
const fm = match[1];
|
||||||
|
const getList = (key: string): string[] => {
|
||||||
|
const m = fm.match(new RegExp(`^${key}:\\s*\\[(.*)\\]$`, 'm'));
|
||||||
|
if (!m || !m[1].trim()) return [];
|
||||||
|
return m[1].split(',').map(s => s.trim().replace(/^"|"$/g, '')).filter(Boolean);
|
||||||
|
};
|
||||||
|
|
||||||
|
return {
|
||||||
|
links: getList('links'),
|
||||||
|
backlinks: getList('backlinks'),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function cosineDistance(a: number[], b: number[]): number {
|
||||||
|
let dot = 0, normA = 0, normB = 0;
|
||||||
|
for (let i = 0; i < a.length; i++) {
|
||||||
|
dot += a[i] * b[i];
|
||||||
|
normA += a[i] * a[i];
|
||||||
|
normB += b[i] * b[i];
|
||||||
|
}
|
||||||
|
const denom = Math.sqrt(normA) * Math.sqrt(normB);
|
||||||
|
return denom === 0 ? 1 : 1 - dot / denom;
|
||||||
|
}
|
||||||
|
|
||||||
|
function getWeekMonday(date: Date = new Date()): string {
|
||||||
|
const d = new Date(Date.UTC(date.getFullYear(), date.getMonth(), date.getDate()));
|
||||||
|
const day = d.getUTCDay();
|
||||||
|
const diff = day === 0 ? -6 : 1 - day;
|
||||||
|
d.setUTCDate(d.getUTCDate() + diff);
|
||||||
|
return d.toISOString().slice(0, 10);
|
||||||
|
}
|
||||||
|
|
||||||
|
function getWeekSunday(monday: string): string {
|
||||||
|
const d = new Date(`${monday}T00:00:00Z`);
|
||||||
|
d.setUTCDate(d.getUTCDate() + 6);
|
||||||
|
return d.toISOString().slice(0, 10);
|
||||||
|
}
|
||||||
|
|
||||||
|
function tagsFromName(name: string): string[] {
|
||||||
|
const prefix = name.split('/')[0];
|
||||||
|
return prefix ? [prefix.toLowerCase()] : [];
|
||||||
|
}
|
||||||
|
|
||||||
|
export function serializeMemory(mem: Memory, week?: {monday: string, sunday: string}): string {
|
||||||
|
return mem.content;
|
||||||
|
}
|
||||||
|
|
||||||
|
export function deserializeMemory(raw: string, embedding: number[] = []): Memory {
|
||||||
|
const match = raw.match(/^---\n([\s\S]*?)\n---\n\n?([\s\S]*)$/);
|
||||||
|
if (!match) {
|
||||||
|
return {name: '', description: '', content: raw.trim(), embedding};
|
||||||
|
}
|
||||||
|
|
||||||
|
const [, fm] = match;
|
||||||
|
const get = (key: string): string => {
|
||||||
|
const m = fm.match(new RegExp(`^${key}:\\s*(.+)$`, 'm'));
|
||||||
|
return m ? m[1].trim() : '';
|
||||||
|
};
|
||||||
|
|
||||||
|
return {
|
||||||
|
name: get('name'),
|
||||||
|
description: get('description'),
|
||||||
|
content: raw.trim(),
|
||||||
|
embedding,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
export class MemoryCache {
|
||||||
|
private tree: KDTree<MemoryRef>;
|
||||||
|
public memories: Memory[];
|
||||||
|
private locks = new Map<string, Promise<void>>();
|
||||||
|
|
||||||
|
constructor(memories: Memory[]) {
|
||||||
|
this.memories = memories;
|
||||||
|
this.tree = this.buildTree();
|
||||||
|
}
|
||||||
|
|
||||||
|
private buildTree(): KDTree<MemoryRef> {
|
||||||
|
const embedded = this.memories.filter(m => m.embedding?.length);
|
||||||
|
if (!embedded.length) return new KDTree<MemoryRef>(0);
|
||||||
|
|
||||||
|
const dims = embedded[0].embedding.length;
|
||||||
|
const points: KDPoint<MemoryRef>[] = embedded.map(m => ({
|
||||||
|
vector: m.embedding,
|
||||||
|
payload: {name: m.name, description: m.description},
|
||||||
|
}));
|
||||||
|
|
||||||
|
return new KDTree<MemoryRef>(dims, 'cosine', points);
|
||||||
|
}
|
||||||
|
|
||||||
|
search(query: number[], limit: number): MemoryRef[] {
|
||||||
|
const results = this.tree.knn(query, limit);
|
||||||
|
return results.map(r => r.point.payload);
|
||||||
|
}
|
||||||
|
|
||||||
|
add(memory: Memory): void {
|
||||||
|
this.memories.push(memory);
|
||||||
|
this.rebuild();
|
||||||
|
}
|
||||||
|
|
||||||
|
update(memory: Memory): void {
|
||||||
|
const idx = this.memories.findIndex(m => m.name === memory.name);
|
||||||
|
if (idx !== -1) {
|
||||||
|
this.memories[idx] = memory;
|
||||||
|
} else {
|
||||||
|
this.memories.push(memory);
|
||||||
|
}
|
||||||
|
this.rebuild();
|
||||||
|
}
|
||||||
|
|
||||||
|
remove(name: string): void {
|
||||||
|
const idx = this.memories.findIndex(m => m.name === name);
|
||||||
|
if (idx !== -1) {
|
||||||
|
this.memories.splice(idx, 1);
|
||||||
|
this.rebuild();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
rebuild(): void {
|
||||||
|
this.tree = this.buildTree();
|
||||||
|
}
|
||||||
|
|
||||||
|
lock<T>(name: string, fn: () => Promise<T>): Promise<T> {
|
||||||
|
const prev = this.locks.get(name) ?? Promise.resolve();
|
||||||
|
let resolveLock!: () => void;
|
||||||
|
const next = new Promise<void>(r => { resolveLock = r; });
|
||||||
|
this.locks.set(name, next);
|
||||||
|
|
||||||
|
const result = prev.then(fn).finally(resolveLock);
|
||||||
|
result.finally(() => {
|
||||||
|
if (this.locks.get(name) === next) this.locks.delete(name);
|
||||||
|
});
|
||||||
|
return result;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
export class MemoryManager {
|
export class MemoryManager {
|
||||||
|
private pendingMemorizations = new Map<string, {
|
||||||
|
memories: Memory[] | MemoryCache,
|
||||||
|
tempMemoryName: string,
|
||||||
|
timestamp: number,
|
||||||
|
}>();
|
||||||
|
|
||||||
tools = {
|
tools = {
|
||||||
edit: (memory: Memory): AiTool => ({
|
read: (memories: Memory[] | MemoryCache): AiTool => ({
|
||||||
name: 'edit_memory',
|
|
||||||
description: 'Edit a memory. Omit start/end to append. Pass start only to replace from that line on (Note line 0 = first line of content / line AFTER description). Pass start+end to replace a specific range. start=0 replaces the whole document. Returns updated document',
|
|
||||||
args: {
|
|
||||||
content: {type: 'string', description: 'New content', required: true},
|
|
||||||
start: {type: 'number', description: 'First line to replace (0-indexed, inclusive). Omit to append.'},
|
|
||||||
end: {type: 'number', description: 'Last line to replace (0-indexed, inclusive). Omit to replace from start to end of doc.'},
|
|
||||||
},
|
|
||||||
fn: (args: any) => {
|
|
||||||
const lines = memory.content ? memory.content.split('\n') : [];
|
|
||||||
const newLines = args.content.split('\n');
|
|
||||||
if(args.start === undefined) lines.push(...newLines);
|
|
||||||
else if(args.end === undefined) lines.splice(args.start, lines.length - args.start, ...newLines);
|
|
||||||
else lines.splice(args.start, args.end - args.start + 1, ...newLines);
|
|
||||||
memory.content = lines.join('\n');
|
|
||||||
return memory.content;
|
|
||||||
}
|
|
||||||
}),
|
|
||||||
extract: (pools: MemoryCollection[]): AiTool => ({
|
|
||||||
name: 'extract_facts',
|
|
||||||
description: 'Extract a list of facts to group into a single memory',
|
|
||||||
args: {
|
|
||||||
name: {type: 'string', description: 'Exact name of an existing memory, or a new name if none fits ([pro]nouns only)', required: true},
|
|
||||||
description: {type: 'string', description: 'One sentence description of the memory subject', required: true},
|
|
||||||
facts: {type: 'string', description: 'Comma separated list of extracted facts', required: true},
|
|
||||||
},
|
|
||||||
fn: (args: any) => {
|
|
||||||
pools.push({
|
|
||||||
name: args.name,
|
|
||||||
description: args.description,
|
|
||||||
facts: args.facts.split(',').map((f: string) => f.trim()).filter(Boolean),
|
|
||||||
});
|
|
||||||
return 'Success';
|
|
||||||
}}),
|
|
||||||
read: (memories: Memory[]): AiTool => ({
|
|
||||||
name: 'read_memory',
|
name: 'read_memory',
|
||||||
description: 'Read entire memory',
|
description: 'Read the full content of a memory document',
|
||||||
args: {
|
args: {
|
||||||
name: {type: 'string', description: 'Exact memory name', required: true},
|
name: {type: 'string', description: 'Exact memory name', required: true},
|
||||||
},
|
},
|
||||||
fn: (args: any) => {
|
fn: (args: any) => {
|
||||||
const mem = memories.find(m => m.name === args.name);
|
const mems = memories instanceof MemoryCache ? memories.memories : memories;
|
||||||
|
const mem = mems.find(m => m.name === args.name);
|
||||||
if (!mem) return 'Document not found';
|
if (!mem) return 'Document not found';
|
||||||
return `Name: ${mem.name}\nDescription: ${mem.description}\n\n${mem.content}`;
|
return mem.content;
|
||||||
}
|
},
|
||||||
}),
|
}),
|
||||||
|
|
||||||
|
forget: (memories: Memory[] | MemoryCache): AiTool => ({
|
||||||
|
name: 'forget_memory',
|
||||||
|
description: 'Permanently delete a memory document and clean up all references to it',
|
||||||
|
args: {
|
||||||
|
name: {type: 'string', description: 'Exact memory name to forget', required: true},
|
||||||
|
reason: {type: 'string', description: 'Why this memory is being deleted', required: true},
|
||||||
|
},
|
||||||
|
fn: (args: any) => {
|
||||||
|
const result = this.forget(args.name, memories);
|
||||||
|
return result ? `Forgotten: ${args.name}` : `Not found: ${args.name}`;
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
|
||||||
|
constructor(private llm: any) {}
|
||||||
|
|
||||||
|
private async createTempMemory(conversation: string): Promise<Memory> {
|
||||||
|
const [e] = await this.llm.embedding(conversation);
|
||||||
|
const timestamp = Date.now();
|
||||||
|
return {
|
||||||
|
name: `_temp_${timestamp}`,
|
||||||
|
description: 'Temporary memory - processing in background',
|
||||||
|
content: `---
|
||||||
|
name: _temp_${timestamp}
|
||||||
|
description: Temporary memory - processing in background
|
||||||
|
tags: [_temporary]
|
||||||
|
links: []
|
||||||
|
backlinks: []
|
||||||
|
modified: ${new Date().toISOString()}
|
||||||
|
---
|
||||||
|
|
||||||
|
# Recent Conversation (Processing)
|
||||||
|
|
||||||
|
${conversation}`,
|
||||||
|
embedding: e?.embedding || [],
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
constructor(private llm: any, private model?: string) {}
|
forget(name: string, memories: Memory[] | MemoryCache): boolean {
|
||||||
|
const mem = memories instanceof MemoryCache ? memories.memories : memories;
|
||||||
|
const idx = mem.findIndex(m => m.name === name);
|
||||||
|
if (idx === -1) return false;
|
||||||
|
|
||||||
/**
|
for (const node of mem) {
|
||||||
* Extracts facts from conversation and groups them into individual memories
|
const {links, backlinks} = extractMetadata(node.content);
|
||||||
* @param {string} conversation Full conversation formatted as [role]: content
|
const newBacklinks = backlinks.filter(b => b !== name);
|
||||||
* @param {Memory[]} memories The user's memory documents
|
const newLinks = links.filter(l => l !== name);
|
||||||
* @param {LLMRequest} options LLM options
|
|
||||||
* @returns {Promise<MemoryCollection[]>} Fact pools grouped by target document
|
|
||||||
*/
|
|
||||||
private async extract(conversation: string, memories: Memory[], options: LLMRequest): Promise<MemoryCollection[]> {
|
|
||||||
const existingDocs = memories.map(m => `Name: ${m.name}\nDescription: ${m.description}`).join('\n\n');
|
|
||||||
const pools: MemoryCollection[] = [];
|
|
||||||
await this.llm.ask(conversation, {
|
|
||||||
model: this.model || options.model,
|
|
||||||
temperature: 0.2,
|
|
||||||
system: `You are a fact extractor. Analyze this conversation and extract facts worth remembering long term.
|
|
||||||
Rules:
|
|
||||||
- ONLY extract facts the USER explicitly stated about themselves or their business
|
|
||||||
- ONLY extract decisions that were MADE during this conversation
|
|
||||||
- DO NOT extract anything the AI said, its name, capabilities, or how it introduced itself
|
|
||||||
- DO NOT extract greetings, pleasantries or generic exchanges
|
|
||||||
- If nothing worth remembering was said, dont do anything, skip calling tools
|
|
||||||
|
|
||||||
For each fact decide whether it belongs in an existing document or needs a new one, then call the \`extract_facts\` tool.
|
if (newBacklinks.length !== backlinks.length || newLinks.length !== links.length) {
|
||||||
|
node.content = this.updateFrontmatter(node.content, {
|
||||||
Existing documents:\n${existingDocs || 'None yet.'}`,
|
links: newLinks,
|
||||||
tools: [this.tools.extract(pools)]
|
backlinks: newBacklinks,
|
||||||
});
|
});
|
||||||
return pools;
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
mem.splice(idx, 1);
|
||||||
* Bot 2 - Editor: merges a pool of facts into a specific document using surgical line-based edits.
|
|
||||||
* Receives full document content and uses read + amend tools to make precise edits.
|
|
||||||
* @param {MemoryCollection} newMem The fact pool to merge
|
|
||||||
* @param {Memory[]} memories The user's memory documents
|
|
||||||
* @param {LLMRequest} options LLM options
|
|
||||||
*/
|
|
||||||
private async edit(newMem: MemoryCollection, memories: Memory[], options: LLMRequest): Promise<void> {
|
|
||||||
const existing = memories.find(m => m.name === newMem.name);
|
|
||||||
const mem: Memory = existing || {name: newMem.name, description: newMem.description || '', content: '', embedding: []};
|
|
||||||
const isNew = !existing;
|
|
||||||
|
|
||||||
await this.llm.ask(newMem.facts.map(f => `- ${f}`).join('\n'),
|
if (memories instanceof MemoryCache) memories.rebuild();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
private cosineSearch(query: number[], memories: Memory[], limit: number): MemoryRef[] {
|
||||||
|
const scored = memories
|
||||||
|
.filter(m => m.embedding?.length)
|
||||||
|
.map(m => ({
|
||||||
|
ref: {name: m.name, description: m.description},
|
||||||
|
distance: cosineDistance(query, m.embedding),
|
||||||
|
}))
|
||||||
|
.sort((a, b) => a.distance - b.distance)
|
||||||
|
.slice(0, limit);
|
||||||
|
return scored.map(s => s.ref);
|
||||||
|
}
|
||||||
|
|
||||||
|
private createNode(name: string, memories: Memory[]): Memory {
|
||||||
|
const existing = memories.find(m => m.name === name);
|
||||||
|
if (existing) return existing;
|
||||||
|
return {
|
||||||
|
name,
|
||||||
|
description: '',
|
||||||
|
content: '',
|
||||||
|
embedding: [],
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
private listNodes(memories: Memory[]): MemoryRef[] {
|
||||||
|
return memories.map(m => ({name: m.name, description: m.description}));
|
||||||
|
}
|
||||||
|
|
||||||
|
async recollect(query: string, memories: Memory[] | MemoryCache, limit = 5, graphDepth = 1): Promise<Memory[]> {
|
||||||
|
const mem: Memory[] = memories instanceof MemoryCache ? memories.memories : memories;
|
||||||
|
if (!mem.length) return [];
|
||||||
|
|
||||||
|
const [e] = await this.llm.embedding(query);
|
||||||
|
if (!e) return [];
|
||||||
|
|
||||||
|
let vectorResults: MemoryRef[];
|
||||||
|
if (memories instanceof MemoryCache) vectorResults = memories.search(e.embedding, limit);
|
||||||
|
else vectorResults = this.cosineSearch(e.embedding, mem, limit);
|
||||||
|
const found = new Set<string>(vectorResults.map(r => r.name));
|
||||||
|
|
||||||
|
if (graphDepth > 0) {
|
||||||
|
const frontier = [...found];
|
||||||
|
for (let depth = 0; depth < graphDepth; depth++) {
|
||||||
|
const next: string[] = [];
|
||||||
|
for (const name of frontier) {
|
||||||
|
const node = mem.find(m => m.name === name);
|
||||||
|
if (!node) continue;
|
||||||
|
const {links} = extractMetadata(node.content);
|
||||||
|
for (const link of links) {
|
||||||
|
if (!found.has(link) && mem.find(m => m.name === link)) {
|
||||||
|
found.add(link);
|
||||||
|
next.push(link);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
frontier.splice(0, frontier.length, ...next);
|
||||||
|
if (!frontier.length) break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const vectorOrder = vectorResults.map(r => r.name);
|
||||||
|
const graphExpansions = [...found].filter(n => !vectorOrder.includes(n));
|
||||||
|
const ordered = [...vectorOrder, ...graphExpansions];
|
||||||
|
return ordered.map(n => mem.find(m => m.name === n)!).filter(Boolean);
|
||||||
|
}
|
||||||
|
|
||||||
|
async memorize(history: LLMMessage[], memories: Memory[] | MemoryCache, options: LLMRequest): Promise<Memory[]> {
|
||||||
|
const conversation = history
|
||||||
|
.filter(h => h.role === 'user' || h.role === 'assistant')
|
||||||
|
.map(h => `[${h.role}]: ${h.content}`).join('\n\n').trim();
|
||||||
|
if(!conversation) return [];
|
||||||
|
|
||||||
|
// Create and insert temp memory immediately
|
||||||
|
const trackingId = `${Date.now()}_${Math.random()}`;
|
||||||
|
const tempMemory = await this.createTempMemory(conversation);
|
||||||
|
const mem = memories instanceof MemoryCache ? memories.memories : memories;
|
||||||
|
mem.push(tempMemory);
|
||||||
|
if (memories instanceof MemoryCache) memories.rebuild();
|
||||||
|
this.pendingMemorizations.set(trackingId, {
|
||||||
|
memories,
|
||||||
|
tempMemoryName: tempMemory.name,
|
||||||
|
timestamp: Date.now(),
|
||||||
|
});
|
||||||
|
|
||||||
|
try {
|
||||||
|
await this._memorizeBackground(conversation, memories, options);
|
||||||
|
// Return the final memories (excluding temp ones)
|
||||||
|
const finalMem = memories instanceof MemoryCache ? memories.memories : memories;
|
||||||
|
return finalMem.filter(m => !m.name.startsWith('_temp_'));
|
||||||
|
} catch (err) {
|
||||||
|
throw err;
|
||||||
|
} finally {
|
||||||
|
// Remove temp memory from the exact same memory array/cache
|
||||||
|
const pending = this.pendingMemorizations.get(trackingId);
|
||||||
|
if (pending) {
|
||||||
|
const cleanMem = pending.memories instanceof MemoryCache
|
||||||
|
? pending.memories.memories
|
||||||
|
: pending.memories;
|
||||||
|
const idx = cleanMem.findIndex(m => m.name === pending.tempMemoryName);
|
||||||
|
if (idx !== -1) {
|
||||||
|
cleanMem.splice(idx, 1);
|
||||||
|
}
|
||||||
|
if (pending.memories instanceof MemoryCache) {
|
||||||
|
pending.memories.rebuild();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
this.pendingMemorizations.delete(trackingId);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private async _memorizeBackground(conversation: string, memories: Memory[] | MemoryCache, options: LLMRequest): Promise<void> {
|
||||||
|
const mem = memories instanceof MemoryCache ? memories.memories : memories;
|
||||||
|
const monday = getWeekMonday();
|
||||||
|
const sunday = getWeekSunday(monday);
|
||||||
|
const buckets = await this.factAgent(conversation, mem, options, monday);
|
||||||
|
if(!buckets.length) return;
|
||||||
|
|
||||||
|
const runDocAgent = (node: Memory, bucket: FactBucket, embedding?: number[], week?: {monday: string, sunday: string}) => {
|
||||||
|
if (memories instanceof MemoryCache) {
|
||||||
|
return memories.lock(node.name, () => this.docAgent(node, bucket, mem, options, embedding, week));
|
||||||
|
}
|
||||||
|
return this.docAgent(node, bucket, mem, options, embedding, week);
|
||||||
|
};
|
||||||
|
|
||||||
|
await Promise.all(buckets.map(async bucket => {
|
||||||
|
let node = mem.find(m => m.name === bucket.subject && !m.name.startsWith('_temp_'));
|
||||||
|
let embedding: number[] | undefined;
|
||||||
|
|
||||||
|
if (!node || bucket.isNew) {
|
||||||
|
const [e] = await this.llm.embedding(`${bucket.subject}\n${bucket.facts.join('\n')}`);
|
||||||
|
embedding = e?.embedding;
|
||||||
|
|
||||||
|
if (!node) {
|
||||||
|
node = this.createNode(bucket.subject, mem);
|
||||||
|
mem.push(node);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const week = bucket.subject.startsWith('Journal/') ? {monday, sunday} : undefined;
|
||||||
|
await runDocAgent(node, bucket, embedding, week);
|
||||||
|
}));
|
||||||
|
|
||||||
|
if(memories instanceof MemoryCache)
|
||||||
|
memories.rebuild();
|
||||||
|
}
|
||||||
|
|
||||||
|
private buildHeader(node: Memory, week?: {monday: string, sunday: string}, links: string[] = [], backlinks: string[] = []): string {
|
||||||
|
const tags = node.name.split('/')[0]?.toLowerCase();
|
||||||
|
const lines = [
|
||||||
|
'---',
|
||||||
|
`name: ${node.name}`,
|
||||||
|
`description: ${node.description || ''}`,
|
||||||
|
tags ? `tags: [${tags}]` : '',
|
||||||
|
links.length ? `links: [${links.map(l => `"${l}"`).join(', ')}]` : 'links: []',
|
||||||
|
backlinks.length ? `backlinks: [${backlinks.map(l => `"${l}"`).join(', ')}]` : 'backlinks: []',
|
||||||
|
week ? `week: ${week.monday} – ${week.sunday}` : '',
|
||||||
|
`modified: ${new Date().toISOString()}`,
|
||||||
|
'---',
|
||||||
|
].filter(Boolean);
|
||||||
|
return lines.join('\n');
|
||||||
|
}
|
||||||
|
|
||||||
|
private applyHeader(content: string, header: string): string {
|
||||||
|
const hasFrontmatter = content.trimStart().startsWith('---');
|
||||||
|
if (hasFrontmatter) {
|
||||||
|
return content.replace(/^---[\s\S]*?---\n?/, `${header}\n`);
|
||||||
|
}
|
||||||
|
return `${header}\n\n${content}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
private updateFrontmatter(content: string, updates: {links?: string[], backlinks?: string[]}): string {
|
||||||
|
const match = content.match(/^---\n([\s\S]*?)\n---\n\n?([\s\S]*)$/);
|
||||||
|
if (!match) return content;
|
||||||
|
|
||||||
|
const [, fm, body] = match;
|
||||||
|
let newFm = fm;
|
||||||
|
|
||||||
|
if (updates.links !== undefined) {
|
||||||
|
const linksList = updates.links.length ? `[${updates.links.map(l => `"${l}"`).join(', ')}]` : '[]';
|
||||||
|
newFm = newFm.replace(/^links:.*$/m, `links: ${linksList}`);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (updates.backlinks !== undefined) {
|
||||||
|
const backlinksList = updates.backlinks.length ? `[${updates.backlinks.map(l => `"${l}"`).join(', ')}]` : '[]';
|
||||||
|
newFm = newFm.replace(/^backlinks:.*$/m, `backlinks: ${backlinksList}`);
|
||||||
|
}
|
||||||
|
|
||||||
|
newFm = newFm.replace(/^modified:.*$/m, `modified: ${new Date().toISOString()}`);
|
||||||
|
|
||||||
|
return `---\n${newFm}\n---\n\n${body}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
private stripHeader(content: string): string {
|
||||||
|
return content.replace(/^---[\s\S]*?---\n?/, '').trimStart();
|
||||||
|
}
|
||||||
|
|
||||||
|
private async docAgent(node: Memory, bucket: FactBucket, memories: Memory[], options: LLMRequest, precomputedEmbedding?: number[], week?: {monday: string, sunday: string}): Promise<void> {
|
||||||
|
const {links: oldLinks} = extractMetadata(node.content);
|
||||||
|
let finalContent = node.content;
|
||||||
|
|
||||||
|
await this.llm.ask(
|
||||||
|
`New facts to integrate:\n${bucket.facts.map(f => `- ${f}`).join('\n')}`,
|
||||||
{
|
{
|
||||||
model: this.model || options.model,
|
model: options.model,
|
||||||
temperature: 0.2,
|
temperature: 0.3,
|
||||||
system: `You are a document editor. Merge the users list of facts into the following document using the \`edit_memory\` tool; call it as many times as necessary:
|
system: `You are a knowledge base editor. Integrate the provided facts into the document below.
|
||||||
\`\`\`
|
|
||||||
${mem.content}
|
Formatting rules:
|
||||||
|
- Use Obsidian-style markdown: # headings, **bold** for key terms, bullet lists for facts
|
||||||
|
- Link related concepts with [[WikiLink]] notation using full paths like [[People/Sarah]] or [[Projects/Website]]
|
||||||
|
- You may create links to nodes that don't exist yet if the concept is important
|
||||||
|
- Keep the document concise, factual, and human-readable
|
||||||
|
- Resolve any contradictions between old content and new facts (new facts win)
|
||||||
|
- Do not add filler, preamble, or AI commentary — just clean knowledge documents
|
||||||
|
- The document begins with a YAML frontmatter block (between --- markers) — do not remove or rewrite it, it is maintained automatically
|
||||||
|
${week ? '- This is a weekly journal entry. The frontmatter contains the week date range.\n' : ''}
|
||||||
|
All nodes:
|
||||||
|
${this.listNodes(memories).map(n => n.name).join(', ') || 'none'}
|
||||||
|
|
||||||
|
Current document:
|
||||||
|
\`\`\`markdown
|
||||||
|
${node.content || '(empty — this is a new document)'}
|
||||||
\`\`\``,
|
\`\`\``,
|
||||||
tools: [this.tools.edit(mem)]
|
tools: [{
|
||||||
|
name: 'update_document',
|
||||||
|
description: 'Write the complete updated document content. Include everything after the frontmatter block — the frontmatter will be recalculated automatically.',
|
||||||
|
args: {
|
||||||
|
description: {type: 'string', description: 'One-line description of what this document covers, no formatting or emojis', required: true},
|
||||||
|
content: {type: 'string', description: 'Document body in markdown, without the frontmatter block', required: true},
|
||||||
|
},
|
||||||
|
fn: (args: any) => {
|
||||||
|
node.description = args.description;
|
||||||
|
finalContent = args.content;
|
||||||
|
return 'Saved';
|
||||||
|
},
|
||||||
|
}],
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
|
|
||||||
if(isNew || mem.description !== existing?.description) {
|
const newLinks = extractLinks(finalContent).filter(l => l !== node.name);
|
||||||
const e = await this.llm.embedding(mem.description);
|
const newLinkSet = new Set(newLinks);
|
||||||
mem.embedding = e?.[0]?.embedding;
|
const oldLinkSet = new Set(oldLinks);
|
||||||
}
|
|
||||||
|
|
||||||
if(isNew) memories.push(mem);
|
for (const added of newLinkSet) {
|
||||||
else {
|
if (!oldLinkSet.has(added)) {
|
||||||
const idx = memories.findIndex(m => m.name === newMem.name);
|
const target = memories.find(m => m.name === added);
|
||||||
if(idx >= 0) memories[idx] = mem;
|
if (target) {
|
||||||
|
const {backlinks} = extractMetadata(target.content);
|
||||||
|
if (!backlinks.includes(node.name)) {
|
||||||
|
target.content = this.updateFrontmatter(target.content, {
|
||||||
|
backlinks: [...backlinks, node.name],
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (const removed of oldLinkSet) {
|
||||||
|
if (!newLinkSet.has(removed)) {
|
||||||
|
const target = memories.find(m => m.name === removed);
|
||||||
|
if (target) {
|
||||||
|
const {backlinks} = extractMetadata(target.content);
|
||||||
|
target.content = this.updateFrontmatter(target.content, {
|
||||||
|
backlinks: backlinks.filter(b => b !== node.name),
|
||||||
|
});
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
const {backlinks} = extractMetadata(node.content);
|
||||||
* Find relevant memory documents for a query using description embeddings
|
const header = this.buildHeader(node, week, newLinks, backlinks);
|
||||||
* @param {string} query The query to search against
|
node.content = this.applyHeader(finalContent, header);
|
||||||
* @param {Memory[]} memories The user's memory documents
|
|
||||||
* @param {number} limit Max number of results to return
|
if (precomputedEmbedding) {
|
||||||
* @returns {Promise<Memory[]>} The most relevant memory documents
|
node.embedding = precomputedEmbedding;
|
||||||
*/
|
} else {
|
||||||
async recollect(query: string, memories: Memory[], limit = 5): Promise<Memory[]> {
|
const embedInput = `${node.description}\n\n${this.stripHeader(node.content)}`.trim();
|
||||||
const [e] = await this.llm.embedding(query);
|
const [e] = await this.llm.embedding(embedInput);
|
||||||
return memories
|
if (e) node.embedding = e.embedding;
|
||||||
.filter(m => m.embedding?.length)
|
}
|
||||||
.map(m => ({...m, score: this.llm.cosineSimilarity(m.embedding, e.embedding)}))
|
|
||||||
.toSorted((a: any, b: any) => b.score - a.score)
|
|
||||||
.slice(0, limit);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
private async factAgent(conversation: string, memories: Memory[], options: LLMRequest, weekKey: string): Promise<FactBucket[]> {
|
||||||
* Two-stage memory pipeline: classify facts from conversation history then surgically merge them into documents.
|
const buckets: FactBucket[] = [];
|
||||||
* Bot 1 (classify) extracts and groups facts cheaply. Bot 2 (edit) runs per-document in parallel with full content access.
|
await this.llm.ask(conversation, {
|
||||||
* @param {LLMMessage[]} history Full conversation history to digest
|
model: options.model,
|
||||||
* @param {Memory[]} memories The user's memory documents — mutated in place
|
temperature: 0.2,
|
||||||
* @param {LLMRequest} options LLM options
|
system: `You are a fact extractor. Analyze this conversation and extract facts worth remembering long-term.
|
||||||
*/
|
|
||||||
async memorize(history: LLMMessage[], memories: Memory[], options: LLMRequest): Promise<void> {
|
Rules:
|
||||||
const conversation = history
|
- ONLY extract facts the USER explicitly stated about themselves, their work, or their projects
|
||||||
.filter(h => h.role === 'user' || h.role === 'assistant')
|
- ONLY extract decisions that were MADE during this conversation
|
||||||
.map(h => `[${h.role}]: ${h.content}`)
|
- DO NOT extract anything the AI said, its capabilities, or meta-conversation about the AI
|
||||||
.join('\n\n');
|
- DO NOT extract greetings, pleasantries, or generic exchanges
|
||||||
if(!conversation.trim()) return;
|
- If nothing worth remembering was said, do not call any tools
|
||||||
const pools = await this.extract(conversation, memories, options);
|
|
||||||
if(!pools.length) return;
|
When extracting facts, you MUST also decide the exact destination path:
|
||||||
await Promise.all(pools.map(pool => this.edit(pool, memories, options)));
|
- Use an existing node name if the facts clearly belong there
|
||||||
|
- All information primary about the user should go under "Personal" (e.g., Personal/Goals, Personal/Habits)
|
||||||
|
- When required, create a new path following collection/subject format (e.g., People/Sarah, Projects/Oxide)
|
||||||
|
- For journal entries, use "journal" (will auto-route to Journal/${weekKey})
|
||||||
|
|
||||||
|
Available nodes:
|
||||||
|
${this.listNodes(memories).map(n => `- ${n.name}: ${n.description}`).join('\n') || 'None yet.'}`,
|
||||||
|
tools: [{
|
||||||
|
name: 'extract_facts',
|
||||||
|
description: 'Submit facts with their destination',
|
||||||
|
args: {
|
||||||
|
destination: {type: 'string', description: 'Exact existing node name OR new path (e.g. "People/Sarah", "Projects/Oxide")', required: true},
|
||||||
|
facts: {type: 'string', description: 'Comma-separated facts', required: true},
|
||||||
|
create_new: {type: 'boolean', description: 'True if this is a new node that doesn\'t exist yet', required: true},
|
||||||
|
},
|
||||||
|
fn: (args: any) => {
|
||||||
|
const subject = args.destination.trim().toLowerCase() === 'journal'
|
||||||
|
? `Journal/${weekKey}`
|
||||||
|
: args.destination;
|
||||||
|
buckets.push({
|
||||||
|
subject,
|
||||||
|
facts: args.facts.split(',').map((f: string) => f.trim()).filter(Boolean),
|
||||||
|
isNew: args.create_new,
|
||||||
|
});
|
||||||
|
return 'Recorded';
|
||||||
|
},
|
||||||
|
}],
|
||||||
|
});
|
||||||
|
return buckets;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
17
src/tools.ts
17
src/tools.ts
@@ -100,16 +100,11 @@ export const CliTool: AiTool = {
|
|||||||
|
|
||||||
export const DateTimeTool: AiTool = {
|
export const DateTimeTool: AiTool = {
|
||||||
name: 'get_datetime',
|
name: 'get_datetime',
|
||||||
description: 'Get local date / time',
|
description: 'Get local/UTC date/time',
|
||||||
args: {},
|
args: {
|
||||||
fn: async () => new Date().toString()
|
timezone: {type: 'string', description: 'Which timezone to return, defaults to local', enum: ['local', 'utc'], default: 'local'}
|
||||||
}
|
},
|
||||||
|
fn: ({timezone}) => new Date()[timezone === 'local' ? 'toString' : 'toUTCString']()
|
||||||
export const DateTimeUTCTool: AiTool = {
|
|
||||||
name: 'get_datetime_utc',
|
|
||||||
description: 'Get current UTC date / time',
|
|
||||||
args: {},
|
|
||||||
fn: async () => new Date().toUTCString()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
export const ExecTool: AiTool = {
|
export const ExecTool: AiTool = {
|
||||||
@@ -168,7 +163,7 @@ export const JSTool: AiTool = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
export const PythonTool: AiTool = {
|
export const PythonTool: AiTool = {
|
||||||
name: 'exec_javascript',
|
name: 'exec_python',
|
||||||
description: 'Execute commonjs javascript',
|
description: 'Execute commonjs javascript',
|
||||||
args: {
|
args: {
|
||||||
code: {type: 'string', description: 'CommonJS javascript', required: true}
|
code: {type: 'string', description: 'CommonJS javascript', required: true}
|
||||||
|
|||||||
Reference in New Issue
Block a user