diff --git a/src/memory.ts b/src/memory.ts index f1c7c28..02579b3 100644 --- a/src/memory.ts +++ b/src/memory.ts @@ -1,95 +1,72 @@ import {MemoryNode, patchGraph, rebuildGraph} from './helpers.ts'; -import {LLMRequest, LLMMessage} from './llm.ts'; -import {AiTool} from './tools.ts'; -import {KDTree} from './kd-tree.ts'; +import {KDTree} from './memory/kd-tree.ts'; +import {Memory, MemoryRef, MemoryStore} from './memory/memory.ts'; +import {cosineDistance, embedMemoryFields} from './utils.ts'; -const FACT_SIMILARITY_THRESHOLD = 0.62; -const PENDING_HEADING = '## Pending'; -const TODO_HEADING = '## Todo list'; const TREE_TOMBSTONE_LIMIT = 0.25; -const ALIAS_MATCH_THRESHOLD = 0.55; -export type Memory = { - name: string; - description: string; - content: string; - embedding: number[]; - titleEmbedding?: number[]; - bodyEmbeddings?: number[][]; - links: string[]; - backlinks: string[]; -} - -type MemoryRef = { - name: string; - description: string; - distance?: number; -} - -type FactBucket = { - subject: string; - facts: string[]; -} - -type MemoryTask = { - /** Exact node name / new persistent entity path this task belongs to, or '' for a personal task with no entity (goes to the journal) */ - subject: string; - task: string; - done: boolean; -} - -type FactAgentResult = { - buckets: FactBucket[]; - journal: string; - tasks: MemoryTask[]; -} - -function dedupeFacts(facts: string[]): string[] { - const seen = new Map(); - for(const f of facts) { - const clean = f.trim(); - if(clean) seen.set(clean.toLowerCase(), clean); +export function memoryStore(memories: MemoryStore): { + list: Memory[]; + cache: MemoryCache | null; + find: (name: string) => Memory | undefined; + ghosts: () => string[]; + search: (vector: number[], limit: number) => MemoryRef[]; + forget: (name: string) => boolean; + rebuild: (changed?: Memory[]) => MemoryNode[]; + backfillEmbeddings: (llm: any) => Promise; +} { + if(memories instanceof MemoryCache) { + return { + list: memories.memories, + cache: memories, + find: name => memories.find(name), + ghosts: () => memories.ghosts(), + search: (vector, limit) => memories.search(vector, limit), + forget: name => memories.remove(name), + rebuild: changed => memories.rebuild(changed), + backfillEmbeddings: llm => memories.backfillEmbeddings(llm), + }; } - return [...seen.values()]; -} -function cosineDistance(a: number[], b: number[]): number { - let dot = 0, normA = 0, normB = 0; - for(let i = 0; i < a.length; i++) { - dot += a[i] * b[i]; - normA += a[i] * a[i]; - normB += b[i] * b[i]; - } - const denom = Math.sqrt(normA) * Math.sqrt(normB); - return denom === 0 ? 1 : 1 - dot / denom; -} + return { + list: memories, + cache: null, + find: name => memories.find(m => m.name === name), + ghosts: () => rebuildGraph(memories).filter(n => n.missing).map(n => n.name), + search: (vector, limit) => memories + .filter(m => m.embedding?.length) + .map(m => ({ + name: m.name, + description: m.description, + distance: cosineDistance(vector, m.embedding), + })) + .sort((a, b) => a.distance - b.distance) + .slice(0, limit), + forget: name => { + const idx = memories.findIndex(m => m.name === name); + if(idx === -1) return false; + memories.splice(idx, 1); + return true; + }, + rebuild: changed => rebuildGraph(memories), + backfillEmbeddings: async llm => { + const missing = memories.filter(m => !m.embedding?.length); + if(!missing.length) return 0; -function cosineSearch(query: number[], memories: Memory[], limit: number): MemoryRef[] { - return memories - .filter(m => m.embedding?.length) - .map(m => ({name: m.name, description: m.description, distance: cosineDistance(query, m.embedding)})) - .sort((a, b) => a.distance - b.distance) - .slice(0, limit); -} + await Promise.all(missing.map(async node => { + const body = node.content.replace(/^---[\s\S]*?\n---\n?/, '').trimStart(); + const [titleE] = await llm.embedding(node.name.split('/').pop() || node.name); + const [descE] = await llm.embedding(node.description || ''); + const bodyChunks = body ? await llm.embedding(body) : []; -async function embedMemoryFields(node: Memory, llm: any): Promise { - const body = stripHeader(node.content); - const [titleE] = await llm.embedding(node.name.split('/').pop() || node.name); - const [descE] = await llm.embedding(node.description || ''); - const bodyChunks = body ? await llm.embedding(body) : []; - if(titleE) node.titleEmbedding = titleE.embedding; - if(descE) node.embedding = descE.embedding; - node.bodyEmbeddings = bodyChunks.map((c: any) => c.embedding).filter(Boolean); -} + if(titleE) node.titleEmbedding = titleE.embedding; + if(descE) node.embedding = descE.embedding; + node.bodyEmbeddings = bodyChunks.map((c: any) => c.embedding).filter(Boolean); + })); -export function stripHeader(content: string): string { - return content.replace(/^---[\s\S]*?\n---\n?/, '').trimStart(); -} - -/** True if a task has no persistent entity of its own and belongs in the journal instead. */ -function isPersonalTask(t: MemoryTask): boolean { - const s = (t.subject ?? '').trim().toLowerCase(); - return !s || s === 'journal' || s.startsWith('journal/'); + return missing.length; + }, + }; } export class MemoryCache { @@ -98,7 +75,9 @@ export class MemoryCache { public memories: Memory[]; public nodes: MemoryNode[] = []; - get length() { return this.memories.length; } + get length() { + return this.memories.length; + } constructor(memories: Memory[]) { this.memories = memories; @@ -106,6 +85,10 @@ export class MemoryCache { this.rebuild(); } + find(name: string): Memory | undefined { + return this.memories.find(m => m.name === name); + } + private syncTree(): void { const current = new Set(this.memories.map(m => m.name)); @@ -120,8 +103,11 @@ export class MemoryCache { for(const mem of this.memories) { if(!mem.embedding?.length || this.indexed.has(mem.name)) continue; if(this.tree.dims === 0) this.tree = new KDTree(mem.embedding.length, 'cosine'); - if(mem.embedding.length !== this.tree.dims) continue; // guard against embedding model/dim drift - this.tree.insert({vector: mem.embedding, payload: {name: mem.name, description: mem.description}}); + if(mem.embedding.length !== this.tree.dims) continue; + this.tree.insert({ + vector: mem.embedding, + payload: {name: mem.name, description: mem.description}, + }); this.indexed.set(mem.name, mem.embedding); } @@ -139,656 +125,41 @@ export class MemoryCache { } update(memory: Memory): void { - const existing = this.memories.find(m => m.name === memory.name); + const existing = this.find(memory.name); if(existing) Object.assign(existing, memory); else this.memories.push(memory); this.rebuild([existing ?? memory]); } - remove(name: string): void { + remove(name: string): boolean { const idx = this.memories.findIndex(m => m.name === name); - if(idx !== -1) { - this.memories.splice(idx, 1); - this.rebuild(); - } - } - - rebuild(changed?: Memory[]): void { - this.nodes = (changed?.length && this.nodes.length) - ? patchGraph(this.memories, this.nodes, changed) - : rebuildGraph(this.memories); - this.syncTree(); - } -} - -class MemoryAccessor { - readonly list: Memory[]; - private readonly cache: MemoryCache | null; - - constructor(memories: Memory[] | MemoryCache) { - this.cache = memories instanceof MemoryCache ? memories : null; - this.list = this.cache ? this.cache.memories : memories; - } - - find(name: string): Memory | undefined { - return this.list.find(m => m.name === name); - } - - commit(changed?: Memory[]): MemoryNode[] { - if(this.cache) { - this.cache.rebuild(changed); - return this.cache.nodes; - } - return rebuildGraph(this.list); - } - - ghosts(): string[] { - const nodes = this.cache ? this.cache.nodes : rebuildGraph(this.list); - return nodes.filter(n => n.missing).map(n => n.name); - } - - search(vector: number[], limit: number): MemoryRef[] { - return this.cache ? this.cache.search(vector, limit) : cosineSearch(vector, this.list, limit); - } - - forget(name: string): boolean { - const idx = this.list.findIndex(m => m.name === name); if(idx === -1) return false; - this.list.splice(idx, 1); - this.commit(); + this.memories.splice(idx, 1); + this.rebuild(); return true; } + ghosts(): string[] { + return this.nodes.filter(n => n.missing).map(n => n.name); + } + + rebuild(changed?: Memory[]): MemoryNode[] { + this.nodes = changed?.length && this.nodes.length + ? patchGraph(this.memories, this.nodes, changed) + : rebuildGraph(this.memories); + this.syncTree(); + return this.nodes; + } + + commit(changed?: Memory[]): MemoryNode[] { + return this.rebuild(changed); + } + async backfillEmbeddings(llm: any): Promise { - const missing = this.list.filter(m => !m.embedding?.length); + const missing = this.memories.filter(m => !m.embedding?.length); if(!missing.length) return 0; await Promise.all(missing.map(node => embedMemoryFields(node, llm))); - this.commit(); + this.commit(missing); return missing.length; } } - -export type MemoryOptions = { - /** Memory object */ - memory: Memory[] | MemoryCache; - /** Inject N memories into the system prompt */ - inject?: boolean; - /** expose recall tool to LLM */ - tool?: boolean; - /** Update memory on compression */ - update?: boolean; - /** Max context size of memories to inject to each call (removed immediately after use) */ - maxTokens?: number; -} - -export class MemoryManager { - private mergeLock: Promise = Promise.resolve(); - private queues = new Map void} | null, - task: Promise, - }>(); - private recentlyTouched = new Map(); - - tools = { - forget: (memories: Memory[] | MemoryCache): AiTool => ({ - name: 'memory_forget', - description: 'Permanently delete a memory document and clean up all references to it', - args: { - name: {type: 'string', description: 'Exact memory name to forget', required: true} - }, - fn: (args: any) => { - const result = this.forget(args.name, memories); - return result ? `Forgotten: ${args.name}` : `Not found: ${args.name}`; - }, - }), - - read: (memories: Memory[] | MemoryCache): AiTool => ({ - name: 'memory_recall', - description: 'Read the full content of a memory document', - args: { - name: {type: 'string', description: 'Exact memory name', required: true} - }, - fn: (args: any) => { - const mem = new MemoryAccessor(memories).find(args.name); - if(!mem) return 'Document not found'; - this.touch(mem.name); - return mem.content; - }, - }), - - search: (memories: Memory[] | MemoryCache): AiTool => ({ - name: 'memory_search', - description: 'Use embeddings to find the MOST relevant memories, even if NOT relevant', - args: { - query: {type: 'string', description: 'What to look for in the memories', required: true}, - limit: {type: 'number', description: 'Number of memories to return', default: 1}, - }, - fn: async ({query, limit}) => { - const mem = await this.recollect(query, memories, limit); - return mem.map(m => `Memory: ${m.name} -Description: ${m.description} -Links: ${[...m.links, ...m.backlinks].join(', ')} -\`\`\` -${m.content} -\`\`\``).join('\n\n'); - }, - }), - }; - - constructor(private llm: any) {} - - static normalize(m?: Memory[] | MemoryCache | MemoryOptions) { - if(!m) return null; - const raw = m instanceof MemoryCache || Array.isArray(m); - return raw ? {memory: m, inject: true, tool: true, update: true} : {inject: true, tool: true, update: true, ...m}; - } - - private stage(node: Memory, block: string): void { - if(!node.content) { - const title = node.name.split('/').pop() ?? node.name; - node.content = this.touchHeader(node, `# ${title}\n`); - } - const body = stripHeader(node.content); - const idx = body.indexOf(PENDING_HEADING); - const newBody = idx === -1 - ? `${body.trimEnd()}\n\n${PENDING_HEADING}\n${block}\n` - : `${body.slice(0, idx + PENDING_HEADING.length)}\n${block}${body.slice(idx + PENDING_HEADING.length)}`; - node.content = this.touchHeader(node, newBody); - } - - private resolveSubject(subject: string, store: MemoryAccessor): string { - function normalize(name: string): string { - return name.trim().toLowerCase().replace(/\s+/g, ' '); - } - - const trimmed = subject.trim(); - const exact = store.find(trimmed); - if(exact) return exact.name; - - const normalized = normalize(trimmed); - const caseInsensitive = store.list.find(m => normalize(m.name) === normalized); - if(caseInsensitive) return caseInsensitive.name; - - const root = trimmed.split('/')[0]; - const leaf = trimmed.split('/').slice(1).join('/') || trimmed; - const candidates = store.list.filter(m => m.name.split('/')[0] === root && m.name !== trimmed); - if(!candidates.length) return trimmed; - - const leaves = candidates.map(m => m.name.split('/').slice(1).join('/') || m.name); - const probe = leaves.length > 1 ? leaves : [...leaves, '']; - const {max, similarities} = this.llm.fuzzyMatch(leaf, ...probe); - if(max >= ALIAS_MATCH_THRESHOLD) return candidates[similarities.indexOf(max)].name; - - return trimmed; - } - - private async factAgent(conversation: string, store: MemoryAccessor, options: LLMRequest): Promise { - const ghosts = store.ghosts(); - - const response = await this.llm.ask(conversation, { - model: options.model, - temperature: 0.2, - system: `Turn this conversation into a persistent memory file by extracting information into organized bullet points - -Think of this like an Obsidian vault with a clear division of responsibility: -- The JOURNAL is a timeline. It answers "what happened, and when" and is the only place with a sense of time. -- ENTITY DOSSIERS are a wiki. They answer "what is currently true about this subject", with no sense of time — only current state. -- Never blur the two: a one-off event, conversation, or debugging session is a journal entry, not an entity, even if it's detailed. - -1. Journal Log -- A chronological, skimmable log of what actually happened: real discussions, decisions made, progress on projects, problems worked through -- This is NOT a transcript, and it is NOT a step-by-step record, its a compressed log of notable events & developments -- One line per development is usually enough: what was worked on and the outcome, not the blow-by-blow of how -- Skip small talk and trivial exchanges entirely. Skip anything that's a todo item (goes in Todo Tasks) or a durable fact about a subject (goes in Entity Dossiers) - -2. Todo Tasks -- Extract concrete tasks the user says need to be done, should be done, or were completed -- Return the task text and whether it is still todo or is done -- A completed task should be marked done, not recreated as a new todo -- Only extract actionable tasks, not general goals or observations, if none - omit returning a tasks array -- Assign each task a subject: - - If the task belongs to a persistent entity (a project, a class, etc.), use that entity's exact node name, or a new entity path if it doesn't exist yet - - If it's a personal/life task with no entity of its own (reach out to someone, reply to an email, pay a bill, etc.), leave subject as an empty string — it belongs in the journal, not a new document - -3. Entity Dossiers -- Detailed dossiers with all factual information regarding a subject -- Record the final/end state, not intermediate changes -- Ignore assistant claims, guesses, greetings, or temporary details -- NEVER create a dossier for something I wouldnt find in a wiki site: temporary information, debugging, guesses, conversations (this is all journal entry stuff!) -- identify its HOME ENTITY: - - The HOME ENTITY name should always be a [abstract|pro]noun - - The grammatical subject/owner of the fact is the strongest clue - - Always preference an existing entity over creating a new one - - New child entities are appropriate only when they are themselves distinct persistent entities - - A document represents a persistent entity, not a topic, feature, bug, event, decision, setting, or conversation fragment - - Put project facts under the project they belong to, person facts under the person, etc - -Example Entity Naming Convention: -- Projects/[Name] -- People/[Name] -- History/[Name] -- Science/[Name] -- [Subject]/[Name] -- Class/[Name]/[Chapter] - -Use [[WikiLinks]] to express relationships between entities. NEVER create documents just to hold relationships -Keep journal material in the journal; don't turn journal events into entities unless they represent something persistent - -Available nodes: -${this.listNodes(store.list).map(n => `- ${n.name}: ${n.description}`).join('\n') || 'None yet.'} -${ghosts.length ? `${ghosts.map(g => `- ${g}: (Ghost)`).join('\n')}` : ''}`, - schema: { - journal: {type: 'string', description: 'Short bullet point recap, omit if nothing notable happened'}, - tasks: { - type: 'array', description: 'Concrete tasks mentioned or completed in the conversation, omit if none', items: { - type: 'object', items: { - subject: {type: 'string', description: 'Exact node name / new persistent entity path this task belongs to, or an empty string if this is a personal task with no entity of its own (those go in the journal)', required: true}, - task: {type: 'string', description: 'Concise actionable task', required: true}, - done: {type: 'boolean', description: 'Whether the task is completed', required: true}, - }, - } - }, - buckets: { - type: 'array', description: 'Groups of facts to remember; omit if none', items: { - type: 'object', items: { - subject: {type: 'string', description: 'Exact node name or new persistent entity path', required: true}, - facts: {type: 'array', description: 'Facts to store here', items: {type: 'string'}}, - }, - }, - }, - }, - }); - - const buckets = new Map(); - for(const bucket of response.buckets ?? []) { - const subject = bucket.subject.trim(); - const facts = buckets.get(subject) ?? []; - facts.push(...dedupeFacts(bucket.facts)); - buckets.set(subject, facts); - } - - return { - buckets: buckets.entries().toArray().map(([subject, facts]) => ({subject, facts})), - journal: (response.journal ?? '').trim(), - tasks: response.tasks ?? [], - }; - } - - private getWeekStart(date: Date = new Date()): string { - const d = new Date(Date.UTC(date.getFullYear(), date.getMonth(), date.getDate())); - const day = d.getUTCDay(); - const diff = day === 0 ? -6 : 1 - day; - d.setUTCDate(d.getUTCDate() + diff); - return d.toISOString().slice(0, 10); - } - - private journalDescription(journalName?: string): string { - const start = journalName?.split('/').pop() || this.getWeekStart(); - const d = new Date(`${start}T00:00:00Z`); - d.setUTCDate(d.getUTCDate() + 6); - const end = d.toISOString().slice(0, 10); - return `Log from ${start} - ${end}`; - } - - private getIncompleteTodos(content: string): string[] { - const body = stripHeader(content); - const match = body.match(/## Todo list\n([\s\S]*?)(?=\n## |$)/i); - if(!match) return []; - return match[1].split('\n') - .map(line => line.match(/^\s*-\s*\[([ xX])\]\s+(.+?)\s*$/)) - .filter((m): m is RegExpMatchArray => !!m && m[1].toLowerCase() !== 'x') - .map(m => m[2].trim()); - } - - private listNodes(memories: Memory[]): MemoryRef[] { - return memories.map(m => ({name: m.name, description: m.description})); - } - - private async mergeAgent(node: Memory, memories: Memory[] | MemoryCache, options: LLMRequest): Promise { - function factSimilarity(a: Memory, b: Memory): number { - if(!a.bodyEmbeddings?.length || !b.bodyEmbeddings?.length) return 0; - let best = 0; - for(const av of a.bodyEmbeddings) { - for(const bv of b.bodyEmbeddings) best = Math.max(best, 1 - cosineDistance(av, bv)); - } - return best; - } - - if(!node.embedding?.length || node.name.startsWith('Journal/')) return null; - const store = new MemoryAccessor(memories); - const candidates = store.list - .filter(m => m.name !== node.name && !m.name.startsWith('Journal/')) - .filter(m => factSimilarity(node, m) >= FACT_SIMILARITY_THRESHOLD); - - if(!candidates.length) return null; - const closest = candidates.sort((a, b) => factSimilarity(node, b) - factSimilarity(node, a))[0]; - const result = await this.llm.ask('', { - model: options.model, - temperature: 0.3, - schema: { - aContent: {type: 'string', description: 'Updated document A body in markdown, without frontmatter.', required: true}, - bContent: {type: 'string', description: 'Updated document B body in markdown, without frontmatter.', required: true}, - }, - system: `Maintain these two persistent knowledge-base documents like a wiki. - -Do NOT merge, rename, or delete either document. Both represent entities that should remain independently addressable. - -The documents were selected because their facts may overlap. Your job is to reconcile duplicated information and connect the documents: -- Decide which document is the HOME for each duplicated fact. -- Keep the authoritative copy in that home document. -- In the other document, replace the information with a short preamble and [[WikiLink]] to the home entity explaining the relationship. -- If the documents are distinct entities but merely related, keep their distinct facts and add useful [[WikiLinks]] between them. -- Do not delete useful entity-specific facts just because they are similar. -- Do not invent relationships or facts. -- Preserve useful history, technical specifics, structure, and existing [[WikiLinks]]. -- Most current truth wins when facts conflict. -- Keep both documents concise and information-dense. -- No frontmatter, preamble, filler, or AI commentary. - -Document A ("${node.name}"): -\`\`\`markdown -${stripHeader(node.content)} -\`\`\` - -Document B ("${closest.name}"): -\`\`\`markdown -${stripHeader(closest.content)} -\`\`\``, - }); - const a = store.find(node.name); - const b = store.find(closest.name); - if(!a || !b || !result?.aContent || !result?.bContent) return null; - a.content = this.touchHeader(a, result.aContent); - b.content = this.touchHeader(b, result.bContent); - await Promise.all([embedMemoryFields(a, this.llm), embedMemoryFields(b, this.llm)]); - return a; - } - - private reconcile(node: Memory, memories: Memory[] | MemoryCache, options: LLMRequest): Promise { - const key = node.name; - const existing = this.queues.get(key); - if(existing) { - existing.dirty = true; - existing.request?.abort?.(); - return existing.task; - } - - const entry = {dirty: false, request: null, task: Promise.resolve()}; - this.queues.set(key, entry); - const store = new MemoryAccessor(memories); - entry.task = (async () => { - let current = node; - try { - do { - entry.dirty = false; - await this.docAgent(current, store.list, options, entry); - this.mergeLock = this.mergeLock.then(() => this.mergeAgent(current, memories, options)); - const result = await this.mergeLock; - if(result) current = result; - } while(entry.dirty); - } finally { - store.commit([node]); - this.queues.delete(key); - } - })(); - return entry.task; - } - - private async docAgent(node: Memory, memories: Memory[], options: LLMRequest, entry: {request: {abort?: () => void} | null}): Promise { - if(!memories.includes(node)) return; - const currentBody = stripHeader(node.content); - const journal = node.name.startsWith('Journal/'); - const system = (journal - ? `You maintain one persistent journal document - -Rewrite the ENTIRE journal, folding "## Pending" into the existing content removing the heading - -Journal design: -- Preserve the chronological daily log -- Maintain a single \`## Todo list\` section for this entity: reconcile tasks semantically (merge equivalent tasks, remove duplicates, preserve incomplete tasks, check off completed ones), and keep it distinct from the narrative/fact sections -- Group information by day under a date heading -- Keep journal entries high level and concise: what was worked on and the outcome, not a step-by-step record of how — that detail lives in conversation history, not here -- Use [[WikiLinks]] for persistent entities; don't turn ordinary journal events into entities -- No frontmatter, preamble, filler, or AI commentary` - : `You maintain one persistent knowledge-base entity document - -Rewrite the ENTIRE document, folding "## Pending" into the existing content. Remove the Pending section when finished. - -Document design: -- The document represents one persistent entity. Keep information about that entity together and organized into sections -- Merge any pending information in, newest fact wins conflicts; remove redundant content -- Maintain a single \`## Todo list\` section for this entity: reconcile tasks semantically (merge equivalent tasks, remove duplicates, preserve incomplete tasks, check off completed ones), and keep it distinct from the narrative/fact sections -- Let the structure fit the entity; there is NO fixed template -- Add headings only when they meaningfully organize recurring information; don't create headings for one-off facts -- Keep the document concise and information-dense without removing useful technical specifics -- Current truth wins when facts conflict. Preserve older conflict as context, only when it adds useful meaning -- No frontmatter, preamble, filler, or AI commentary`) + ` - -Available nodes to link to: -${this.listNodes(memories).filter(n => n.name !== node.name).map(n => n.name).join(', ') || 'none'} - -Current document: -\`\`\`markdown -${currentBody} -\`\`\``; - let update; - try { - for(let i = 0; i < 2 && !update?.content; i++) { - const request = this.llm.ask(currentBody, { - model: options.model, - temperature: 0.3, - schema: { - description: {type: 'string', description: 'One factual sentence describing the document\'s ENTIRE SUBJECT MATTER — for use as a search/merge fingerprint', required: true}, - content: {type: 'string', description: 'Rewritten document body in markdown, without the frontmatter block', required: true}, - }, - system, - }); - entry.request = request; - update = await request; - } - } catch(err: any) { - if(err?.name === 'AbortError') return; - throw err; - } finally { - entry.request = null; - } - - if(!update?.content) return; - node.description = node.name.startsWith('Journal/') ? this.journalDescription(node.name) : node.name !== 'People/User' ? update.description.replaceAll(/[\n:]/g, '') : 'All information about the current user'; - node.content = this.touchHeader(node, update.content); - await embedMemoryFields(node, this.llm); - } - - private parseFrontmatter(content: string): {fm: Map, body: string} { - const match = content.match(/^---\n([\s\S]*?)\n---\n?([\s\S]*)$/); - if(!match) return {fm: new Map(), body: content}; - const fm = new Map(); - for(const line of match[1].split('\n')) { - const i = line.indexOf(':'); - if(i === -1) continue; - const key = line.slice(0, i).trim(); - const raw = line.slice(i + 1).trim(); - let value = raw; - try { value = JSON.parse(raw); } catch { } - fm.set(key, value); - } - return {fm, body: match[2]}; - } - - private touchHeader(node: Memory, body: string): string { - const {fm} = this.parseFrontmatter(node.content); - fm.set('name', node.name); - fm.set('description', (node.name.startsWith('Journal/') ? this.journalDescription(node.name) : node.description) || 'Persistent memory document'); - fm.set('modified', new Date().toISOString()); - return this.writeFrontmatter(fm, stripHeader(body)); - } - - private writeFrontmatter(fm: Map, body: string): string { - const lines = [...fm.entries()].map(([k, v]) => `${k}: ${JSON.stringify(String(v).replace(/\s+/g, ' ').trim())}`); - return `---\n${lines.join('\n')}\n---\n\n${body.trimStart()}`; - } - - decay() { - for(const [name, ttl] of this.recentlyTouched) { - if(ttl <= 1) this.recentlyTouched.delete(name); - else this.recentlyTouched.set(name, ttl - 1); - } - } - - touch(name: string, ttl = 2) { - this.recentlyTouched.set(name, ttl); - } - - forget(name: string, memories: Memory[] | MemoryCache): boolean { - return new MemoryAccessor(memories).forget(name); - } - - async recollect(query: string, memories: Memory[] | MemoryCache, limit = 5, graphDepth = 1): Promise { - function rank(query: number[], candidates: Memory[], limit: number): Memory[] { - const scored = candidates.map(m => { - const titleSim = m.titleEmbedding?.length ? 1 - cosineDistance(query, m.titleEmbedding) : 0; - const descSim = m.embedding?.length ? 1 - cosineDistance(query, m.embedding) : 0; - const bodySim = m.bodyEmbeddings?.length - ? Math.max(...m.bodyEmbeddings.map(b => 1 - cosineDistance(query, b))) - : 0; - return {memory: m, score: titleSim * 0.5 + descSim * 0.35 + bodySim * 0.15}; - }); - return scored.sort((a, b) => b.score - a.score).slice(0, limit).map(s => s.memory); - } - - const store = new MemoryAccessor(memories); - if(!store.list.length) return []; - await store.backfillEmbeddings(this.llm); - - const [e] = await this.llm.embedding(query); - if(!e) return []; - - const pool = store.search(e.embedding, Math.max(limit * 3, limit)); - const poolMemories = pool.map(r => store.find(r.name)).filter((m): m is Memory => !!m); - const ranked = rank(e.embedding, poolMemories, limit); - const found = new Set(ranked.map(m => m.name)); - - if(graphDepth > 0) { - let frontier = [...found]; - for(let depth = 0; depth < graphDepth && frontier.length; depth++) { - const next: string[] = []; - for(const name of frontier) { - const node = store.find(name); - if(!node) continue; - for(const link of node.links) { - if(!found.has(link) && store.find(link)) { - found.add(link); - next.push(link); - } - } - } - frontier = next; - } - } - - const rankedOrder = ranked.map(m => m.name); - const graphExpansions = [...found].filter(n => !rankedOrder.includes(n)); - return [...rankedOrder, ...graphExpansions].map(n => store.find(n)!).filter(Boolean); - } - - async memorize(history: LLMMessage[], memories: Memory[] | MemoryCache, options: LLMRequest): Promise { - const conversation = history - .filter(h => h.role === 'user' || h.role === 'assistant') - .map(h => `[${h.role}]: ${h.content}`).join('\n\n').trim(); - if(!conversation) return []; - - const uid = `${Date.now()}_${Math.random().toString(36).slice(2)}`; - const pending = {role: 'tool', name: 'memory_process', id: uid, content: conversation} as unknown as LLMMessage; - history.push(pending); - - const store = new MemoryAccessor(memories); - const {buckets, journal, tasks} = await this.factAgent(conversation, store, options); - const touched: Memory[] = []; - - const personalTasks = tasks.filter(isPersonalTask); - const entityTasks = tasks.filter(t => !isPersonalTask(t)); - - if(journal || personalTasks.length) { - const journalName = `Journal/${this.getWeekStart()}`; - let jnode = store.find(journalName); - const isNew = !jnode; - if(!jnode) { - jnode = { - name: journalName, - description: this.journalDescription(), - content: '', - embedding: [], - links: [], - backlinks: [], - }; - store.list.push(jnode); - } - - const blocks: string[] = []; - if(journal) blocks.push(`### ${new Date().toISOString().slice(0, 10)}\n${journal}`); - if(isNew) { - const previousDate = new Date(`${this.getWeekStart()}T00:00:00Z`); - previousDate.setUTCDate(previousDate.getUTCDate() - 7); - const previous = store.find(`Journal/${previousDate.toISOString().slice(0, 10)}`); - if(previous) { - const todos = this.getIncompleteTodos(previous.content); - if(todos.length) blocks.push(`${TODO_HEADING}\n${todos.map(task => `- [ ] ${task}`).join('\n')}`); - } - } - if(personalTasks.length) blocks.push(`${TODO_HEADING}\n${personalTasks.map(task => `- [${task.done ? 'x' : ' '}] ${task.task}`).join('\n')}`); - if(blocks.length) this.stage(jnode, blocks.join('\n\n')); - touched.push(jnode); - } - - const entityStaging = new Map(); - for(const {subject, facts} of buckets) { - const resolved = this.resolveSubject(subject, store); - const entry = entityStaging.get(resolved) ?? {facts: [], tasks: []}; - entry.facts.push(...facts); - entityStaging.set(resolved, entry); - } - for(const task of entityTasks) { - const resolved = this.resolveSubject(task.subject, store); - const entry = entityStaging.get(resolved) ?? {facts: [], tasks: []}; - entry.tasks.push(task); - entityStaging.set(resolved, entry); - } - - for(const [resolved, {facts, tasks: subjectTasks}] of entityStaging) { - let node = store.find(resolved); - if(!node) { - node = {name: resolved, description: 'Persistent memory document', content: '', embedding: [], links: [], backlinks: []}; - store.list.push(node); - } - const blocks: string[] = []; - if(facts.length) blocks.push(facts.map(f => `- ${f}`).join('\n')); - if(subjectTasks.length) blocks.push(`${TODO_HEADING}\n${subjectTasks.map(t => `- [${t.done ? 'x' : ' '}] ${t.task}`).join('\n')}`); - if(blocks.length) this.stage(node, blocks.join('\n\n')); - touched.push(node); - } - - await Promise.all(touched.map(async node => { - await embedMemoryFields(node, this.llm); - this.touch(node.name); - })); - - if(touched.length) { - store.commit(touched); - (pending as any).content = `Saved to ${touched.map(n => `[[${n.name}]]`).join(', ')}`; - Promise.all(touched.map(node => this.reconcile(node, memories, options).catch(() => {}))); - } else { - (pending as any).content = 'Nothing worth remembering.'; - } - - (touched as any).uid = uid; - return touched; - } - - async reconcileAll(memories: Memory[] | MemoryCache, options: LLMRequest, scope: 'touched' | 'all' = 'touched'): Promise { - const store = new MemoryAccessor(memories); - const targets = scope === 'all' ? store.list : store.list.filter(m => m.content.includes(PENDING_HEADING)); - await Promise.all(targets.map(node => this.reconcile(node, memories, options))); - store.commit(); - } -} diff --git a/src/memory/doc-agent.ts b/src/memory/doc-agent.ts new file mode 100644 index 0000000..cfcab22 --- /dev/null +++ b/src/memory/doc-agent.ts @@ -0,0 +1,111 @@ +import {memoryStore} from '../memory.ts'; +import {Memory, MemoryStore} from './memory.ts'; +import {LLMRequest} from '../llm.ts'; +import {embedMemoryFields, journalDescription, stripHeader, touchHeader} from '../utils.ts'; + +export const PENDING_HEADING = '## Pending'; +export const TODO_HEADING = '## Todo List'; + +export async function docAgent(llm: any, node: Memory, memories: MemoryStore, options: LLMRequest, entry?: {request: {abort?: () => void} | null},): Promise { + const store = memoryStore(memories); + if(!store.list.includes(node)) return; + + const currentBody = stripHeader(node.content); + const journal = node.name.startsWith('Journal/'); + + const system = `You maintain a Markdown ${journal ? 'journal' : 'wiki file'}, produce the final document body + +- Incorporate the \`${PENDING_HEADING}\` into the wiki & remove it. These new facts trump conflicting information +- Preserve existing structure, wording and context unless directly affected by \`${PENDING_HEADING}\` information +- Only restructure if clearly unorganized, or there is duplicate, stale or misleading information. Do not restructure merely because you dont like it +- Let the structure fit the document; do not force a template +- Use headings, subheadings, lists, tables, code blocks and other formatting where useful +- Create [[WikiLinks]] to existing pages or ghost nodes when the relationship is meaningful but does not exist yet +- Remove duplication and obsolete information, preserve useful detail +- Do not invent facts or make unnecessary changes +- Never add edit commentary` + (journal ? ` +- This is a JOURNAL document, focus on creating a chronological report of events +- Keep events chronological and concise. Do not reorganize events into +- subject-based wiki sections. Preserve dates and useful temporal context. + +Rough Template: +\`\`\`markdown +# Journal + +## ${TODO_HEADING} +- [ ] ... + +### YYYY-MM-DD +- ... +\`\`\`` : ` +- Use only sections that have content and make sense +- Keep related facts together +- historical context may be preserved +- Use ## Related for meaningful links to other wiki pages. + +Rough Template: +# Title + +## ${TODO_HEADING} +- [ ] ... + +## Overview +... + +## Heading +... + +### Subheading +... + +## Related +- [[wikiLinks]]: relationship description +`) + ` + +Available pages for wikilinks: +${store.list.filter(n => n.name !== node.name).map(n => n.name).join(', ') || 'none'}`; + + let update; + + try { + for(let i = 0; i < 2 && !update?.content; i++) { + const request = llm.ask(currentBody, { + model: options.model, + temperature: 0.3, + schema: { + description: { + type: 'string', + description: 'One factual sentence describing the document\'s ENTIRE SUBJECT MATTER — for use as a search/merge fingerprint', + required: true, + }, + content: { + type: 'string', + description: 'Rewritten document body in markdown, without the frontmatter block', + required: true, + }, + }, + system, + }); + + if(entry) entry.request = request; + update = await request; + } + } catch(err: any) { + if(err?.name === 'AbortError') return; + throw err; + } finally { + if(entry) entry.request = null; + } + + if(!update?.content) return; + + node.description = node.name.startsWith('Journal/') + ? journalDescription(node.name) + : node.name !== 'People/User' + ? update.description.replaceAll(/[\n:]/g, '') + : 'All information about the current user'; + + node.content = touchHeader(node, update.content); + + await embedMemoryFields(node, llm); +} diff --git a/src/memory/fact-agent.ts b/src/memory/fact-agent.ts new file mode 100644 index 0000000..6ec9342 --- /dev/null +++ b/src/memory/fact-agent.ts @@ -0,0 +1,154 @@ +import {MemoryCache, memoryStore} from '../memory.ts'; +import {LLMRequest} from '../llm.ts'; +import {Memory, MemoryStore} from './memory.ts'; + +const ALIAS_MATCH_THRESHOLD = 0.55; + +export type FactBucket = { + subject: string; + facts: string[]; +} + +export type MemoryTask = { + subject: string; + task: string; + done: boolean; +} + +export type FactAgentResult = { + buckets: FactBucket[]; + journal: string; + tasks: MemoryTask[]; +} + +function dedupeFacts(facts: string[]): string[] { + const seen = new Map(); + for(const f of facts) { + const clean = f.trim(); + if(clean) seen.set(clean.toLowerCase(), clean); + } + return [...seen.values()]; +} + +export function resolveSubject(subject: string, memories: Memory[] | MemoryCache, llm: any,): string { + function normalize(name: string): string { + return name.trim().toLowerCase().replace(/\s+/g, ' '); + } + + const list = memories instanceof MemoryCache ? memories.memories : memories; + + const trimmed = subject.trim(); + const exact = list.find(m => m.name === trimmed); + if(exact) return exact.name; + + const normalized = normalize(trimmed); + const caseInsensitive = list.find(m => normalize(m.name) === normalized); + if(caseInsensitive) return caseInsensitive.name; + + const root = trimmed.split('/')[0]; + const leaf = trimmed.split('/').slice(1).join('/') || trimmed; + const candidates = list.filter(m => m.name.split('/')[0] === root && m.name !== trimmed); + if(!candidates.length) return trimmed; + + const leaves = candidates.map(m => m.name.split('/').slice(1).join('/') || m.name); + const probe = leaves.length > 1 ? leaves : [...leaves, '']; + const {max, similarities} = llm.fuzzyMatch(leaf, ...probe); + + if(max >= ALIAS_MATCH_THRESHOLD) return candidates[similarities.indexOf(max)].name; + + return trimmed; +} + +export async function factAgent(llm: any, conversation: string, memories: MemoryStore, options: LLMRequest,): Promise { + const store = memoryStore(memories); + const ghosts = store.ghosts(); + + const response = await llm.ask(conversation, { + model: options.model, + temperature: 0.2, + system: `Turn this conversation into a persistent memory file by extracting information into organized bullet points + +Think of this like an Obsidian vault with a clear division of responsibility: +- The JOURNAL is a chronological log. It answers "what happened, and when" and is the only place with a sense of time. +- ENTITY DOSSIERS are a wiki pages. They answer "what is currently true about this subject", with no sense of time — only current state. +- Never blur the two: a one-off event, conversation, or debugging session is a journal entry, not an entity, even if it's detailed. +- Never extract this assistant's own tools, capabilities, or system behavior — that's not memory, it's spec. +- Skip greetings, pleasantries, small talk and generic exchanges entirely. + +1. Journal Log +- A chronological, skimmable log of what actually happened: high level discussions, decisions made (including ones reached jointly with the assistant), progress on projects, problems worked through +- This is NOT a transcript, and it is NOT a step-by-step record, its a compressed day-to-day log of notable events & developments +- One line per development is usually enough: what was worked on and the outcome, not the blow-by-blow of how +- Skip anything that's a todo item (goes in Todo Tasks) or a durable fact about a subject (goes in Entity Dossiers) + +2. Todo Tasks +- Extract concrete tasks the user says need to be done, should be done, or were completed +- Return the task text and whether it is still todo or is done +- A completed task should be marked done, not recreated as a new todo +- Only extract actionable tasks, not general goals or observations, if none - omit returning a tasks array +- Assign each task a HOME ENTITY subject using the same rules as Entity Dossiers below, or an empty string if it belongs in the journal: personal/life task (reach out to someone, pay a bill, etc.) + +3. Entity Dossiers +- Detailed dossiers with all factual information regarding a subject +- Only extract facts the USER explicitly stated about themselves, their work, projects, or decisions made during this conversation — not assistant claims, guesses, or temporary/debugging details +- Record the final/end state, not intermediate deltas +- NEVER create a dossier for something I wouldn't find on a wiki: temporary info, debugging, guesses, conversation fragments (that's journal material) + +Entity Dossier Routing — every fact belongs to exactly one HOME ENTITY: the [pro]noun that OWNS the fact, not the initializer +- All facts primarily about the user go under People/User +- Path format is always Collection/Subject (People/Sarah, Projects/Oxide) — never a bare/rootless name +- ALWAYS prefer an existing node (including ghosts) over creating a new one, even under a different alias — route the facts there, and record the unused name as a ghost node +- Only use child paths (Collection/Subject/Aspect) when a clear child/parent relationship exists between entities +- Use [[WikiLinks]] to express relationships between entities. Ghost links are supported so NEVER create a document just to hold a relationship + +Example Entity Naming Convention: +- Projects/[Name] +- People/[Name] +- History/[Name] +- Science/[Name] +- [Subject]/[Name] +- Class/[Name]/[Chapter] + +Available nodes: +${store.list.map(n => `- ${n.name}: ${n.description}`).join('\n') || 'None yet.'} +${ghosts.length ? `${ghosts.map(g => `- ${g}: (Ghost)`).join('\n')}` : ''}`, + schema: { + journal: { + type: 'string', + description: 'Short bullet point recap, omit if nothing notable happened', + }, + tasks: { + type: 'array', + description: 'Concrete tasks mentioned or completed in the conversation, omit if none', + items: {type: 'object', items: { + subject: {type: 'string', description: 'Exact node name / new persistent entity path this task belongs to, or an empty string if this is a personal task with no entity of its own (those go in the journal)', required: true,}, + task: {type: 'string', description: 'Concise actionable task', required: true,}, + done: {type: 'boolean', description: 'Whether the task is completed', required: true,}, + }}, + }, + buckets: { + type: 'array', + description: 'Groups of facts to remember; omit if none', + items: {type: 'object', items: { + subject: {type: 'string', description: 'Exact node name or new persistent entity path', required: true,}, + facts: {type: 'array', description: 'Facts to store here', items: {type: 'string'},}, + }}, + }, + }, + }); + + const buckets = new Map(); + + for(const bucket of response.buckets ?? []) { + const subject = bucket.subject.trim(); + const facts = buckets.get(subject) ?? []; + facts.push(...dedupeFacts(bucket.facts)); + buckets.set(subject, facts); + } + + return { + buckets: buckets.entries().toArray().map(([subject, facts]) => ({subject, facts})), + journal: (response.journal ?? '').trim(), + tasks: response.tasks ?? [], + }; +} diff --git a/src/kd-tree.ts b/src/memory/kd-tree.ts similarity index 91% rename from src/kd-tree.ts rename to src/memory/kd-tree.ts index 7efc99f..bfa3ae0 100644 --- a/src/kd-tree.ts +++ b/src/memory/kd-tree.ts @@ -1,3 +1,5 @@ +import {cosineDistance, euclideanDistance} from '../utils.ts'; + export type DistanceMetric = "euclidean" | "cosine"; export interface KDPoint { @@ -18,28 +20,6 @@ interface KDNode { deleted?: boolean; } -// ─── Distance helpers ───────────────────────────────────────────────────────── - -function euclidean(a: number[], b: number[]): number { - let sum = 0; - for (let i = 0; i < a.length; i++) { - const d = a[i] - b[i]; - sum += d * d; - } - return Math.sqrt(sum); -} - -function cosine(a: number[], b: number[]): number { - let dot = 0, normA = 0, normB = 0; - for (let i = 0; i < a.length; i++) { - dot += a[i] * b[i]; - normA += a[i] * a[i]; - normB += b[i] * b[i]; - } - const denom = Math.sqrt(normA) * Math.sqrt(normB); - return denom === 0 ? 1 : 1 - dot / denom; // distance = 1 - similarity -} - /** * Keeps the k closest candidates in memory, evicts the furthest when full */ @@ -123,7 +103,7 @@ export class KDTree { points?: KDPoint[] ) { this.dims = dims; - this.distanceFn = metric === "cosine" ? cosine : euclidean; + this.distanceFn = metric === "cosine" ? cosineDistance : euclideanDistance; if (points && points.length > 0) { this.validateAll(points); @@ -300,11 +280,8 @@ export class KDTree { : [node.right, node.left]; this.searchKNN(near, query, k, heap, depth + 1); - - // Only explore the far side if it could contain a closer point. - // For cosine distance we can't prune by axis gap alone, so always explore. const shouldExplore = - this.distanceFn === cosine + this.distanceFn === cosineDistance ? true : Math.abs(diff) < heap.worstDistance; @@ -340,7 +317,7 @@ export class KDTree { this.searchRadius(near, query, radius, results, depth + 1); const shouldExplore = - this.distanceFn === cosine ? true : Math.abs(diff) <= radius; + this.distanceFn === cosineDistance ? true : Math.abs(diff) <= radius; if (shouldExplore) { this.searchRadius(far, query, radius, results, depth + 1); diff --git a/src/memory/librarian-agent.ts b/src/memory/librarian-agent.ts new file mode 100644 index 0000000..e340286 --- /dev/null +++ b/src/memory/librarian-agent.ts @@ -0,0 +1,79 @@ +import {memoryStore} from '../memory.ts'; +import {Memory, MemoryStore} from './memory.ts'; +import {LLMRequest} from '../llm.ts'; + +export type LibrarianAction = + | {type: 'merge'; source: string; target: string; instructions?: string} + | {type: 'split'; target: string; documents: {name: string; instructions?: string}[]} + | {type: 'move'; source: string; target: string} + | {type: 'delete'; target: string} + | {type: 'link'; source: string; target: string}; + +export type LibrarianResult = { + actions: LibrarianAction[]; +}; + +export async function librarianAgent(llm: any, node: Memory, memories: MemoryStore, options: LLMRequest, limit = 8,): Promise { + if(node.name.startsWith('Journal/')) return {actions: []}; + const store = memoryStore(memories); + const candidates = store.search(node.embedding, limit + 1) + .filter(result => result.name !== node.name && !result.name.startsWith('Journal/')) + .map(result => store.find(result.name)) + .filter((memory): memory is Memory => !!memory); + + if(!candidates.length) return {actions: []}; + const documents = [node, ...candidates]; + const response = await llm.ask('', { + model: options.model, + temperature: 0.2, + schema: { + actions: { + type: 'array', + description: 'Actions needed to improve the organization of the knowledge base. Omit if no changes are needed.', + items: {type: 'object', items: { + type: {type: 'string', description: 'One of: merge, split, move, delete, link', required: true,}, + source: {type: 'string', description: 'Source document name. Required for merge, move, and link.',}, + target: {type: 'string', description: 'Target document name. Required for merge, move, delete, and link.',}, + instructions: {type: 'string', description: 'Specific instructions for performing the action.',}, + documents: {type: 'array', description: 'Documents to create when splitting a document.', items: {type: 'object', items: { + name: {type: 'string', required: true}, + instructions: {type: 'string'}, + }}}}, + }, + }, + }, + system: `You are a librarian of a wiki. You keep it organized, coherent, and easy to navigate. You DONT rewrite pages but rather, recommend actions to organize the wiki + +Available actions: +- merge: Two pages represent the same entity and should be merged together +- split: One page contains significant information on two distinct entities and should become separate pages +- move: A document belongs under a different name or path for consistancy +- delete: A document is obsolete, accidental or contains nothing redeamable +- link: Two distinct entities are meaningfully related and should reference each other + +Rules: +- Only act on high confidence +- Semantic similarity alone is NOT sufficient reason to merge two distinct entities +- Prefer preserving existing entities and paths when possible +- ALWAYS use & preserve [[wikilinks]] (including ghost nodes and journal entries) to represent relationships +- A document should represent one persistent entity, not a temporal state, feature, bug, event, decision, or conversation +- When merging, use the more common name as the destination and use ghost nodes to reference old aliases +- When splitting, each resulting document must represent a distinct persistent entity +- Return an empty actions array when the documents are already organized correctly + +Documents being reviewed: + +${documents.map(memory => `### ${memory.name} +Description: ${memory.description} + +Links: ${[...memory.links, ...memory.backlinks].join(', ') || 'none'} + +\`\`\`markdown +${memory.content} +\`\`\``).join('\n\n')}`, + }); + + return { + actions: response.actions ?? [], + }; +} diff --git a/src/memory/memory.ts b/src/memory/memory.ts new file mode 100644 index 0000000..90a8792 --- /dev/null +++ b/src/memory/memory.ts @@ -0,0 +1,28 @@ +import {MemoryCache} from '../memory.ts'; + +export type Memory = { + name: string; + description: string; + content: string; + embedding: number[]; + titleEmbedding?: number[]; + bodyEmbeddings?: number[][]; + links: string[]; + backlinks: string[]; +} + +export type MemoryRef = { + name: string; + description: string; + distance?: number; +} + +export type MemoryOptions = { + memory: Memory[] | MemoryCache; + inject?: boolean; + tool?: boolean; + update?: boolean; + maxTokens?: number; +} + +export type MemoryStore = Memory[] | MemoryCache; diff --git a/src/utils.ts b/src/utils.ts new file mode 100644 index 0000000..86c7aa2 --- /dev/null +++ b/src/utils.ts @@ -0,0 +1,82 @@ +import {Memory} from './memory/memory.ts'; + +export function cosineDistance(a: number[], b: number[]): number { + let dot = 0, normA = 0, normB = 0; + for(let i = 0; i < a.length; i++) { + dot += a[i] * b[i]; + normA += a[i] * a[i]; + normB += b[i] * b[i]; + } + const denom = Math.sqrt(normA) * Math.sqrt(normB); + return denom === 0 ? 1 : 1 - dot / denom; +} + +export async function embedMemoryFields(node: Memory, llm: any): Promise { + const body = stripHeader(node.content); + const [titleE] = await llm.embedding(node.name.split('/').pop() || node.name); + const [descE] = await llm.embedding(node.description || ''); + const bodyChunks = body ? await llm.embedding(body) : []; + if(titleE) node.titleEmbedding = titleE.embedding; + if(descE) node.embedding = descE.embedding; + node.bodyEmbeddings = bodyChunks.map((c: any) => c.embedding).filter(Boolean); +} + +export function euclideanDistance(a: number[], b: number[]): number { + let sum = 0; + for(let i = 0; i < a.length; i++) { + const d = a[i] - b[i]; + sum += d * d; + } + return Math.sqrt(sum); +} + +function getWeekStart(date: Date = new Date()): string { + const d = new Date(Date.UTC(date.getFullYear(), date.getMonth(), date.getDate())); + const day = d.getUTCDay(); + const diff = day === 0 ? -6 : 1 - day; + d.setUTCDate(d.getUTCDate() + diff); + return d.toISOString().slice(0, 10); +} + +export function journalDescription(journalName?: string): string { + const start = journalName?.split('/').pop() || getWeekStart(); + const d = new Date(`${start}T00:00:00Z`); + d.setUTCDate(d.getUTCDate() + 6); + const end = d.toISOString().slice(0, 10); + return `Log from ${start} - ${end}`; +} + +function parseFrontmatter(content: string): {fm: Map, body: string} { + const match = content.match(/^---\n([\s\S]*?)\n---\n?([\s\S]*)$/); + if(!match) return {fm: new Map(), body: content}; + const fm = new Map(); + for(const line of match[1].split('\n')) { + const i = line.indexOf(':'); + if(i === -1) continue; + const key = line.slice(0, i).trim(); + const raw = line.slice(i + 1).trim(); + let value = raw; + try { value = JSON.parse(raw); } catch { } + fm.set(key, value); + } + return {fm, body: match[2]}; +} + +export function writeFrontmatter(fm: Map, body: string): string { + const lines = [...fm.entries()].map(([k, v]) => + `${k}: ${JSON.stringify(String(v).replace(/\s+/g, ' ').trim())}`); + return `---\n${lines.join('\n')}\n---\n\n${body.trimStart()}`; +} + +export function stripHeader(content: string): string { + return content.replace(/^---[\s\S]*?\n---\n?/, '').trimStart(); +} + +export function touchHeader(node: Memory, body: string): string { + const {fm} = parseFrontmatter(node.content); + fm.set('name', node.name); + fm.set('description', (node.name.startsWith('Journal/') ? journalDescription(node.name) : node.description) + || 'Persistent memory document'); + fm.set('modified', new Date().toISOString()); + return writeFrontmatter(fm, stripHeader(body)); +}