|
|
|
|
@@ -1,10 +1,13 @@
|
|
|
|
|
import {MemoryNode, rebuildGraph} from './helpers.ts';
|
|
|
|
|
import {MemoryNode, patchGraph, rebuildGraph} from './helpers.ts';
|
|
|
|
|
import {LLMRequest, LLMMessage} from './llm.ts';
|
|
|
|
|
import {AiTool} from './tools.ts';
|
|
|
|
|
import {KDPoint, KDTree} from './kd-tree.ts';
|
|
|
|
|
|
|
|
|
|
const FACTS_HEADING = '## Facts';
|
|
|
|
|
import {KDTree} from './kd-tree.ts';
|
|
|
|
|
import {escapeRegex} from '@ztimson/utils';
|
|
|
|
|
|
|
|
|
|
const MERGE_THRESHOLD = 0.12;
|
|
|
|
|
const PENDING_HEADING = '## Pending';
|
|
|
|
|
const TREE_TOMBSTONE_LIMIT = 0.25;
|
|
|
|
|
const ALIAS_MATCH_THRESHOLD = 0.55;
|
|
|
|
|
const GENERIC_TEMPLATE = `# {{Title}}
|
|
|
|
|
|
|
|
|
|
## Summary
|
|
|
|
|
@@ -17,7 +20,12 @@ export type Memory = {
|
|
|
|
|
name: string;
|
|
|
|
|
description: string;
|
|
|
|
|
content: string;
|
|
|
|
|
/** Description embedding — indexed in the KD tree, used for merge/ANN candidate lookup */
|
|
|
|
|
embedding: number[];
|
|
|
|
|
/** Title-only embedding, weighted heaviest during recall ranking */
|
|
|
|
|
titleEmbedding?: number[];
|
|
|
|
|
/** Chunked body embeddings, best-chunk match used during recall ranking */
|
|
|
|
|
bodyEmbeddings?: number[][];
|
|
|
|
|
links: string[];
|
|
|
|
|
backlinks: string[];
|
|
|
|
|
}
|
|
|
|
|
@@ -25,6 +33,8 @@ export type Memory = {
|
|
|
|
|
type MemoryRef = {
|
|
|
|
|
name: string;
|
|
|
|
|
description: string;
|
|
|
|
|
/** Cosine distance from the query, present when returned from a search */
|
|
|
|
|
distance?: number;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
type FactBucket = {
|
|
|
|
|
@@ -32,6 +42,11 @@ type FactBucket = {
|
|
|
|
|
facts: string[];
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
type FactAgentResult = {
|
|
|
|
|
buckets: FactBucket[];
|
|
|
|
|
journal: string;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
function dedupeFacts(facts: string[]): string[] {
|
|
|
|
|
const seen = new Map<string, string>();
|
|
|
|
|
for (const f of facts) {
|
|
|
|
|
@@ -55,10 +70,20 @@ function cosineDistance(a: number[], b: number[]): number {
|
|
|
|
|
function cosineSearch(query: number[], memories: Memory[], limit: number): MemoryRef[] {
|
|
|
|
|
return memories
|
|
|
|
|
.filter(m => m.embedding?.length)
|
|
|
|
|
.map(m => ({ref: {name: m.name, description: m.description}, distance: cosineDistance(query, m.embedding)}))
|
|
|
|
|
.map(m => ({name: m.name, description: m.description, distance: cosineDistance(query, m.embedding)}))
|
|
|
|
|
.sort((a, b) => a.distance - b.distance)
|
|
|
|
|
.slice(0, limit)
|
|
|
|
|
.map(s => s.ref);
|
|
|
|
|
.slice(0, limit);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Re-embed a node's title / description / body fields. Description embedding stays the KD-tree index key. */
|
|
|
|
|
async function embedMemoryFields(node: Memory, llm: any): Promise<void> {
|
|
|
|
|
const body = stripHeader(node.content);
|
|
|
|
|
const [titleE] = await llm.embedding(node.name.split('/').pop() || node.name);
|
|
|
|
|
const [descE] = await llm.embedding(node.description || '');
|
|
|
|
|
const bodyChunks = body ? await llm.embedding(body) : [];
|
|
|
|
|
if (titleE) node.titleEmbedding = titleE.embedding;
|
|
|
|
|
if (descE) node.embedding = descE.embedding;
|
|
|
|
|
node.bodyEmbeddings = bodyChunks.map((c: any) => c.embedding).filter(Boolean);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
export function stripHeader(content: string): string {
|
|
|
|
|
@@ -67,6 +92,8 @@ export function stripHeader(content: string): string {
|
|
|
|
|
|
|
|
|
|
export class MemoryCache {
|
|
|
|
|
private tree!: KDTree<MemoryRef>;
|
|
|
|
|
/** Tracks which memories are currently indexed in the tree, keyed by name -> embedding reference */
|
|
|
|
|
private indexed = new Map<string, number[]>();
|
|
|
|
|
public memories: Memory[];
|
|
|
|
|
public nodes: MemoryNode[] = [];
|
|
|
|
|
|
|
|
|
|
@@ -74,37 +101,48 @@ export class MemoryCache {
|
|
|
|
|
|
|
|
|
|
constructor(memories: Memory[]) {
|
|
|
|
|
this.memories = memories;
|
|
|
|
|
this.tree = new KDTree<MemoryRef>(0);
|
|
|
|
|
this.rebuild();
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private buildTree(): KDTree<MemoryRef> {
|
|
|
|
|
const embedded = this.memories.filter(m => m.embedding?.length);
|
|
|
|
|
if (!embedded.length) return new KDTree<MemoryRef>(0);
|
|
|
|
|
/** Incrementally sync the KD tree against `this.memories` instead of rebuilding from scratch */
|
|
|
|
|
private syncTree(): void {
|
|
|
|
|
const current = new Set(this.memories.map(m => m.name));
|
|
|
|
|
|
|
|
|
|
const dims = embedded[0].embedding.length;
|
|
|
|
|
const points: KDPoint<MemoryRef>[] = embedded.map(m => ({
|
|
|
|
|
vector: m.embedding,
|
|
|
|
|
payload: {name: m.name, description: m.description},
|
|
|
|
|
}));
|
|
|
|
|
for (const [name, emb] of [...this.indexed]) {
|
|
|
|
|
const mem = this.memories.find(m => m.name === name);
|
|
|
|
|
if (!mem || !current.has(name) || mem.embedding !== emb) {
|
|
|
|
|
this.tree.remove(p => p.name === name);
|
|
|
|
|
this.indexed.delete(name);
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return new KDTree<MemoryRef>(dims, 'cosine', points);
|
|
|
|
|
for (const mem of this.memories) {
|
|
|
|
|
if (!mem.embedding?.length || this.indexed.has(mem.name)) continue;
|
|
|
|
|
if (this.tree.dims === 0) this.tree = new KDTree<MemoryRef>(mem.embedding.length, 'cosine');
|
|
|
|
|
if (mem.embedding.length !== this.tree.dims) continue; // guard against embedding model/dim drift
|
|
|
|
|
this.tree.insert({vector: mem.embedding, payload: {name: mem.name, description: mem.description}});
|
|
|
|
|
this.indexed.set(mem.name, mem.embedding);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (this.tree.tombstoneRatio > TREE_TOMBSTONE_LIMIT) this.tree.rebalance();
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
search(query: number[], limit: number): MemoryRef[] {
|
|
|
|
|
if (!this.tree || this.tree.dims === 0) return [];
|
|
|
|
|
return this.tree.knn(query, limit).map(r => r.point.payload);
|
|
|
|
|
return this.tree.knn(query, limit).map(r => ({...r.point.payload, distance: r.distance}));
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
add(memory: Memory): void {
|
|
|
|
|
this.memories.push(memory);
|
|
|
|
|
this.rebuild();
|
|
|
|
|
this.rebuild([memory]);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
update(memory: Memory): void {
|
|
|
|
|
const idx = this.memories.findIndex(m => m.name === memory.name);
|
|
|
|
|
if (idx !== -1) this.memories[idx] = memory;
|
|
|
|
|
const existing = this.memories.find(m => m.name === memory.name);
|
|
|
|
|
if (existing) Object.assign(existing, memory);
|
|
|
|
|
else this.memories.push(memory);
|
|
|
|
|
this.rebuild();
|
|
|
|
|
this.rebuild([existing ?? memory]);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
remove(name: string): void {
|
|
|
|
|
@@ -115,9 +153,11 @@ export class MemoryCache {
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
rebuild(): void {
|
|
|
|
|
this.nodes = rebuildGraph(this.memories);
|
|
|
|
|
this.tree = this.buildTree();
|
|
|
|
|
rebuild(changed?: Memory[]): void {
|
|
|
|
|
this.nodes = (changed?.length && this.nodes.length)
|
|
|
|
|
? patchGraph(this.memories, this.nodes, changed)
|
|
|
|
|
: rebuildGraph(this.memories);
|
|
|
|
|
this.syncTree();
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
@@ -134,9 +174,9 @@ class MemoryAccessor {
|
|
|
|
|
return this.list.find(m => m.name === name);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
commit(): MemoryNode[] {
|
|
|
|
|
commit(changed?: Memory[]): MemoryNode[] {
|
|
|
|
|
if (this.cache) {
|
|
|
|
|
this.cache.rebuild();
|
|
|
|
|
this.cache.rebuild(changed);
|
|
|
|
|
return this.cache.nodes;
|
|
|
|
|
}
|
|
|
|
|
return rebuildGraph(this.list);
|
|
|
|
|
@@ -147,6 +187,7 @@ class MemoryAccessor {
|
|
|
|
|
return nodes.filter(n => n.missing).map(n => n.name);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Cache path uses the KD tree's knn(); raw-array path (no cache available) falls back to a linear cosine scan */
|
|
|
|
|
search(vector: number[], limit: number): MemoryRef[] {
|
|
|
|
|
return this.cache ? this.cache.search(vector, limit) : cosineSearch(vector, this.list, limit);
|
|
|
|
|
}
|
|
|
|
|
@@ -162,10 +203,7 @@ class MemoryAccessor {
|
|
|
|
|
async backfillEmbeddings(llm: any): Promise<number> {
|
|
|
|
|
const missing = this.list.filter(m => !m.embedding?.length);
|
|
|
|
|
if (!missing.length) return 0;
|
|
|
|
|
await Promise.all(missing.map(async node => {
|
|
|
|
|
const [e] = await llm.embedding(node.content);
|
|
|
|
|
if (e) node.embedding = e.embedding;
|
|
|
|
|
}));
|
|
|
|
|
await Promise.all(missing.map(node => embedMemoryFields(node, llm)));
|
|
|
|
|
this.commit();
|
|
|
|
|
return missing.length;
|
|
|
|
|
}
|
|
|
|
|
@@ -185,13 +223,13 @@ export type MemoryOptions = {
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
export class MemoryManager {
|
|
|
|
|
private recentlyTouched = new Map<string, number>();
|
|
|
|
|
|
|
|
|
|
private mergeLock: Promise<any> = Promise.resolve();
|
|
|
|
|
private queues = new Map<string, {
|
|
|
|
|
dirty: boolean,
|
|
|
|
|
request: {abort?: () => void} | null,
|
|
|
|
|
task: Promise<void>,
|
|
|
|
|
}>();
|
|
|
|
|
private recentlyTouched = new Map<string, number>();
|
|
|
|
|
|
|
|
|
|
tools = {
|
|
|
|
|
forget: (memories: Memory[] | MemoryCache): AiTool => ({
|
|
|
|
|
@@ -251,14 +289,13 @@ ${m.content}
|
|
|
|
|
return new MemoryAccessor(memories);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private appendFacts(node: Memory, facts: string[]): void {
|
|
|
|
|
private stage(node: Memory, block: string): void {
|
|
|
|
|
this.ensureDoc(node);
|
|
|
|
|
const body = stripHeader(node.content);
|
|
|
|
|
const bullets = facts.map(f => `- ${f}`).join('\n');
|
|
|
|
|
const idx = body.indexOf(FACTS_HEADING);
|
|
|
|
|
const idx = body.indexOf(PENDING_HEADING);
|
|
|
|
|
const newBody = idx === -1
|
|
|
|
|
? `${body.trimEnd()}\n\n${FACTS_HEADING}\n${bullets}\n`
|
|
|
|
|
: `${body.slice(0, idx + FACTS_HEADING.length)}\n${bullets}${body.slice(idx + FACTS_HEADING.length)}`;
|
|
|
|
|
? `${body.trimEnd()}\n\n${PENDING_HEADING}\n${block}\n`
|
|
|
|
|
: `${body.slice(0, idx + PENDING_HEADING.length)}\n${block}${body.slice(idx + PENDING_HEADING.length)}`;
|
|
|
|
|
node.content = this.touchHeader(node, newBody);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
@@ -268,42 +305,93 @@ ${m.content}
|
|
|
|
|
node.content = this.touchHeader(node, `# ${title}\n`);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private async factAgent(conversation: string, store: MemoryAccessor, options: LLMRequest, weekKey: string): Promise<FactBucket[]> {
|
|
|
|
|
private sanitizeDescription(text: string): string {
|
|
|
|
|
return (text ?? '').replace(/\s+/g, ' ').trim().slice(0, 240);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private relink(memories: Memory[], from: string, to: string): void {
|
|
|
|
|
const pattern = new RegExp(`\\[\\[${escapeRegex(from)}\\]\\]`, 'g');
|
|
|
|
|
for (const m of memories) if (pattern.test(m.content)) m.content = m.content.replace(pattern, `[[${to}]]`);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private normalizeLeaf(name: string): string {
|
|
|
|
|
return name.trim().toLowerCase().replace(/\s+/g, ' ');
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Resolve a fact-agent proposed subject to an existing node when it's an alias/rename of one.
|
|
|
|
|
* Exact match is checked first (cheap, and covers the common case since node names are
|
|
|
|
|
* already normalized at creation time). Only falls through to fuzzy alias matching against
|
|
|
|
|
* same-root candidates when there's no existing hit — i.e. only on likely-new-doc creation.
|
|
|
|
|
*/
|
|
|
|
|
private resolveSubject(subject: string, store: MemoryAccessor): string {
|
|
|
|
|
const trimmed = subject.trim();
|
|
|
|
|
const exact = store.find(trimmed);
|
|
|
|
|
if (exact) return exact.name;
|
|
|
|
|
|
|
|
|
|
const normalized = this.normalizeLeaf(trimmed);
|
|
|
|
|
const caseInsensitive = store.list.find(m => this.normalizeLeaf(m.name) === normalized);
|
|
|
|
|
if (caseInsensitive) return caseInsensitive.name;
|
|
|
|
|
|
|
|
|
|
const root = trimmed.split('/')[0];
|
|
|
|
|
const leaf = trimmed.split('/').slice(1).join('/') || trimmed;
|
|
|
|
|
const candidates = store.list.filter(m => m.name.split('/')[0] === root && m.name !== trimmed);
|
|
|
|
|
if (!candidates.length) return trimmed;
|
|
|
|
|
|
|
|
|
|
// fuzzyMatch requires >=2 terms; pad with an empty string when there's only one candidate
|
|
|
|
|
const leaves = candidates.map(m => m.name.split('/').slice(1).join('/') || m.name);
|
|
|
|
|
const probe = leaves.length > 1 ? leaves : [...leaves, ''];
|
|
|
|
|
const {max, similarities} = this.llm.fuzzyMatch(leaf, ...probe);
|
|
|
|
|
if (max >= ALIAS_MATCH_THRESHOLD) return candidates[similarities.indexOf(max)].name;
|
|
|
|
|
|
|
|
|
|
return trimmed;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private async factAgent(conversation: string, store: MemoryAccessor, options: LLMRequest): Promise<FactAgentResult> {
|
|
|
|
|
const ghosts = store.ghosts();
|
|
|
|
|
|
|
|
|
|
const response = await this.llm.ask(conversation, {
|
|
|
|
|
model: options.model,
|
|
|
|
|
temperature: 0.2,
|
|
|
|
|
system: `You are a fact extractor to build obsidian knowledge vaults.
|
|
|
|
|
Analyze this conversation and extract facts worth remembering long-term.
|
|
|
|
|
system: `You are a fact extractor for Obsidian-style knowledge vaults. Analyze the conversation and produce:
|
|
|
|
|
|
|
|
|
|
Rules:
|
|
|
|
|
- Always extract facts that the user explicitly told you to remember
|
|
|
|
|
- ONLY extract current facts the USER explicitly stated about themselves, their work, projects or decisions that were MADE during this conversation
|
|
|
|
|
- DO NOT extract greetings, pleasantries, or generic exchanges
|
|
|
|
|
- DO NOT extract deltas or changes in facts; ONLY the end fact
|
|
|
|
|
- DO NOT extract anything the AI/assistant itself said
|
|
|
|
|
- If nothing worth remembering was said, return an empty buckets array
|
|
|
|
|
1. Journal recap (single paragraph)
|
|
|
|
|
- "Captains Log" style record keeping
|
|
|
|
|
- What was discussed/worked on, decisions, user's events/state/mood, general context
|
|
|
|
|
- Leave empty only for trivial/empty exchanges/small talk
|
|
|
|
|
|
|
|
|
|
When extracting facts, you MUST also decide the exact destination path:
|
|
|
|
|
- Reuse node names (including ghost) as much as possible IF the facts belongs there
|
|
|
|
|
- All information primarily about the user should go under "People/User"
|
|
|
|
|
- When required, create a new path following collection/subject format (e.g., People/Sarah, Projects/Oxide) — you are not limited to any fixed list of collections, use whatever fits
|
|
|
|
|
- For journal entries, use "Journal"
|
|
|
|
|
2. Fact buckets
|
|
|
|
|
- ONLY facts the USER explicitly stated about themselves, their work, projects, or decisions made during this conversation
|
|
|
|
|
- NEVER extract greetings, pleasantries, or anything the assistant itself said
|
|
|
|
|
- Extract the final/end state, not deltas
|
|
|
|
|
|
|
|
|
|
Path assignment (entity) rules:
|
|
|
|
|
- Use the owning entity of the fact (even if implied): "New bug on project 51 -> Projects/51"
|
|
|
|
|
- When multiple facts relate to the same entity, pick a primary owner and wikilink related entities
|
|
|
|
|
- Reuse existing entities when the owner already has a node
|
|
|
|
|
- Always group under consistent entity roots (always plural):
|
|
|
|
|
- Projects/[Name] for all initiatives
|
|
|
|
|
- People/[Name] for all individuals
|
|
|
|
|
- History/[Name] for all historical figures/events
|
|
|
|
|
- Science/[Name] for all scientific concepts
|
|
|
|
|
- Child entities nest under their parent entity:
|
|
|
|
|
- Projects/51/Memory System, Projects/51/Bug-XYZ, not Bugs/51
|
|
|
|
|
- Science/AI/Model-X, not Model-X/AI
|
|
|
|
|
|
|
|
|
|
Wikilink rules:
|
|
|
|
|
- Use [[WikiLinks]] to connect related entities (e.g., [[Projects/51]], [[People/Robert]])
|
|
|
|
|
- Only link specific, existing or implied entity paths — skip generic terms
|
|
|
|
|
- Don't over-link: each link should add clarity or context, not noise
|
|
|
|
|
|
|
|
|
|
Available nodes:
|
|
|
|
|
- Journal
|
|
|
|
|
${this.listNodes(store.list).filter(n => !n.name.includes('Journal')).map(n => `- ${n.name}: ${n.description}`).join('\n') || 'None yet.'}
|
|
|
|
|
${this.listNodes(store.list).map(n => `- ${n.name}: ${n.description}`).join('\n') || 'None yet.'}
|
|
|
|
|
${ghosts.length ? `${ghosts.map(g => `- ${g}: (Ghost)`).join('\n')}` : ''}`,
|
|
|
|
|
schema: {
|
|
|
|
|
buckets: {type: 'array', description: 'Groups of facts to remember, each assigned to a different node. Return an empty array if there is nothing worth storing in an obsidian vault', items: {
|
|
|
|
|
type: 'object', items: {
|
|
|
|
|
subject: {type: 'string', description: 'Exact existing node name OR new path (e.g. "People/Sarah", "Projects/Oxide"), or "Journal"', required: true},
|
|
|
|
|
facts: {
|
|
|
|
|
type: 'array',
|
|
|
|
|
description: 'Facts to store at this destination',
|
|
|
|
|
items: {type: 'string', description: 'A single fact'},
|
|
|
|
|
},
|
|
|
|
|
journal: {type: 'string', description: 'Short day-to-day recap; empty if nothing happened.', required: false},
|
|
|
|
|
buckets: {type: 'array', description: 'Groups of facts to remember; empty array if nothing worth storing.', items: {
|
|
|
|
|
type: 'object', items: {
|
|
|
|
|
subject: {type: 'string', description: 'Exact node name or new path (e.g. "People/Sarah", "Projects/Oxide")', required: true},
|
|
|
|
|
facts: {type: 'array', description: 'Facts to store here', items: {type: 'string'}},
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
@@ -311,15 +399,17 @@ ${ghosts.length ? `${ghosts.map(g => `- ${g}: (Ghost)`).join('\n')}` : ''}`,
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
const buckets = new Map<string, string[]>();
|
|
|
|
|
for(const bucket of response.buckets ?? []) {
|
|
|
|
|
const subject = bucket.subject.trim().toLowerCase() === 'journal'
|
|
|
|
|
? `Journal/${weekKey}` : bucket.subject.trim();
|
|
|
|
|
for (const bucket of response.buckets ?? []) {
|
|
|
|
|
const subject = bucket.subject.trim();
|
|
|
|
|
const facts = buckets.get(subject) ?? [];
|
|
|
|
|
facts.push(...dedupeFacts(bucket.facts));
|
|
|
|
|
buckets.set(subject, facts);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return buckets.entries().toArray().map(([subject, facts]) => ({subject, facts}));
|
|
|
|
|
return {
|
|
|
|
|
buckets: buckets.entries().toArray().map(([subject, facts]) => ({subject, facts})),
|
|
|
|
|
journal: (response.journal ?? '').trim(),
|
|
|
|
|
};
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private getWeekMonday(date: Date = new Date()): string {
|
|
|
|
|
@@ -334,6 +424,36 @@ ${ghosts.length ? `${ghosts.map(g => `- ${g}: (Ghost)`).join('\n')}` : ''}`,
|
|
|
|
|
return memories.map(m => ({name: m.name, description: m.description}));
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Finds the closest merge candidate via the KD tree's knn() instead of a manual O(n) cosine scan */
|
|
|
|
|
private async checkMerge(node: Memory, memories: Memory[] | MemoryCache, options: LLMRequest, threshold = MERGE_THRESHOLD): Promise<Memory | null> {
|
|
|
|
|
if (!node.embedding?.length || node.name.startsWith('Journal/')) return null;
|
|
|
|
|
const store = this.access(memories);
|
|
|
|
|
|
|
|
|
|
const candidate = store.search(node.embedding, 5)
|
|
|
|
|
.find(r => r.name !== node.name && !r.name.startsWith('Journal/') && r.distance !== undefined && r.distance <= threshold);
|
|
|
|
|
if (!candidate) return null;
|
|
|
|
|
const closest = store.find(candidate.name);
|
|
|
|
|
if (!closest) return null;
|
|
|
|
|
|
|
|
|
|
const result = await this.mergeAgent(node, closest, options);
|
|
|
|
|
const merged: Memory = {name: result.name, description: this.sanitizeDescription(result.description), content: '', embedding: [], links: [], backlinks: []};
|
|
|
|
|
merged.content = this.touchHeader(merged, result.content);
|
|
|
|
|
await embedMemoryFields(merged, this.llm);
|
|
|
|
|
|
|
|
|
|
this.relink(store.list, node.name, merged.name);
|
|
|
|
|
this.relink(store.list, closest.name, merged.name);
|
|
|
|
|
|
|
|
|
|
this.queues.get(closest.name)?.request?.abort?.();
|
|
|
|
|
this.queues.delete(closest.name);
|
|
|
|
|
|
|
|
|
|
store.forget(node.name);
|
|
|
|
|
store.forget(closest.name);
|
|
|
|
|
store.list.push(merged);
|
|
|
|
|
store.commit();
|
|
|
|
|
|
|
|
|
|
return merged;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private reconcile(node: Memory, memories: Memory[] | MemoryCache, options: LLMRequest): Promise<void> {
|
|
|
|
|
const key = node.name;
|
|
|
|
|
const existing = this.queues.get(key);
|
|
|
|
|
@@ -347,18 +467,25 @@ ${ghosts.length ? `${ghosts.map(g => `- ${g}: (Ghost)`).join('\n')}` : ''}`,
|
|
|
|
|
this.queues.set(key, entry);
|
|
|
|
|
const store = this.access(memories);
|
|
|
|
|
entry.task = (async () => {
|
|
|
|
|
do {
|
|
|
|
|
entry.dirty = false;
|
|
|
|
|
await this.docAgent(node, store.list, options, entry);
|
|
|
|
|
} while (entry.dirty);
|
|
|
|
|
})().finally(() => {
|
|
|
|
|
this.queues.delete(key);
|
|
|
|
|
store.commit();
|
|
|
|
|
});
|
|
|
|
|
let current = node, merged = false;
|
|
|
|
|
try {
|
|
|
|
|
do {
|
|
|
|
|
entry.dirty = false;
|
|
|
|
|
await this.docAgent(current, store.list, options, entry);
|
|
|
|
|
this.mergeLock = this.mergeLock.then(() => this.checkMerge(current, memories, options));
|
|
|
|
|
const result = await this.mergeLock;
|
|
|
|
|
if (result) { current = result; merged = true; }
|
|
|
|
|
} while (entry.dirty);
|
|
|
|
|
} finally {
|
|
|
|
|
store.commit(merged ? undefined : [node]);
|
|
|
|
|
this.queues.delete(key);
|
|
|
|
|
}
|
|
|
|
|
})();
|
|
|
|
|
return entry.task;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private async docAgent(node: Memory, memories: Memory[], options: LLMRequest, entry: {request: {abort?: () => void} | null}): Promise<void> {
|
|
|
|
|
if(!memories.includes(node)) return;
|
|
|
|
|
const currentBody = stripHeader(node.content);
|
|
|
|
|
let update;
|
|
|
|
|
try {
|
|
|
|
|
@@ -367,27 +494,27 @@ ${ghosts.length ? `${ghosts.map(g => `- ${g}: (Ghost)`).join('\n')}` : ''}`,
|
|
|
|
|
model: options.model,
|
|
|
|
|
temperature: 0.3,
|
|
|
|
|
schema: {
|
|
|
|
|
description: {type: 'string', description: 'One-line description of what this document covers, no formatting or emojis', required: true},
|
|
|
|
|
description: {type: 'string', description: 'One factual sentence describing the document\'s ENTIRE SUBJECT MATTER — for use as a search/merge fingerprint', required: true},
|
|
|
|
|
content: {type: 'string', description: 'Rewritten document body in markdown, without the frontmatter block', required: true},
|
|
|
|
|
},
|
|
|
|
|
system: `You are a knowledge base editor maintaining one document in an Obsidian-style vault.
|
|
|
|
|
system: `You are a knowledge base editor maintaining one Obsidian-style document.
|
|
|
|
|
|
|
|
|
|
If the document has a "${FACTS_HEADING}" section, integrate every bullet under it into the appropriate part of the document, then remove the "${FACTS_HEADING}" section entirely. If there is no such section, just tidy the document per the rules below.
|
|
|
|
|
If it has a "## Pending" section, fold all new material into the appropriate part, resolve overlap, then remove the section entirely. If no section, just tidy per the rules below.
|
|
|
|
|
|
|
|
|
|
Structure: follow this generic shape loosely, adapting section names/order to what the content actually needs (e.g. journal-style docs may want a timeline instead of "Details"):
|
|
|
|
|
Use this loose structure, adapting headings to what the content needs:
|
|
|
|
|
\`\`\`markdown
|
|
|
|
|
${GENERIC_TEMPLATE}
|
|
|
|
|
\`\`\`
|
|
|
|
|
|
|
|
|
|
Formatting rules:
|
|
|
|
|
- Use Obsidian-style markdown: # headings, **bold** for emphasis, bullet & numbered lists for grouped 1D data, tables for 2D data
|
|
|
|
|
- Link related concepts with [[WikiLink]] notation using full paths like [[People/Sarah]] or [[Projects/Website]]
|
|
|
|
|
- Create links for specific entities (person, place, project, program) and abstract concepts, but skip generics (car, red, dog)
|
|
|
|
|
- Keep the document concise, factual, and human-readable
|
|
|
|
|
- Resolve contradictions: newer facts always win — delete the outdated statement entirely, never keep both
|
|
|
|
|
- Do not add frontmatter blocks, filler, preamble, or AI commentary
|
|
|
|
|
Rules:
|
|
|
|
|
- Contradictions: "## Pending" holds the newest information — bias toward it. Fold it in as the standing fact and drop the outdated statement, unless the old context adds meaningful nuance (e.g. "previously X, now Y"). This document should read as a source of truth, not an audit log
|
|
|
|
|
- Journals (Journal/...): keep entries as a chronological timeline; clean up grammar within entries but never delete history
|
|
|
|
|
- Use Obsidian markdown: # headings, **bold**, bullet/numbered lists, tables for 2D data
|
|
|
|
|
- Link specific entities and concepts with [[WikiLink]] (e.g., [[Projects/KiwixServer]]); skip generics
|
|
|
|
|
- Keep concise, factual, human-readable
|
|
|
|
|
- NO frontmatter, filler, preamble, or AI commentary
|
|
|
|
|
|
|
|
|
|
Other nodes in the vault (link to these instead of duplicating their content):
|
|
|
|
|
Available nodes to link to (don't duplicate their content):
|
|
|
|
|
${this.listNodes(memories).filter(n => n.name !== node.name).map(n => n.name).join(', ') || 'none'}
|
|
|
|
|
|
|
|
|
|
Current document:
|
|
|
|
|
@@ -406,10 +533,41 @@ ${currentBody}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (!update?.content) return;
|
|
|
|
|
node.description = node.name !== 'People/User' ? update.description : 'All information about the current user';
|
|
|
|
|
node.description = node.name !== 'People/User' ? this.sanitizeDescription(update.description) : 'All information about the current user';
|
|
|
|
|
node.content = this.touchHeader(node, update.content);
|
|
|
|
|
const [e] = await this.llm.embedding(node.content);
|
|
|
|
|
if (e) node.embedding = e.embedding;
|
|
|
|
|
await embedMemoryFields(node, this.llm);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private async mergeAgent(a: Memory, b: Memory, options: LLMRequest): Promise<{name: string, description: string, content: string}> {
|
|
|
|
|
const modifiedOf = (m: Memory) => this.parseFrontmatter(m.content).fm.get('modified') || 'unknown';
|
|
|
|
|
|
|
|
|
|
return this.llm.ask('', {
|
|
|
|
|
model: options.model,
|
|
|
|
|
temperature: 0.3,
|
|
|
|
|
schema: {
|
|
|
|
|
name: {type: 'string', description: 'New path for the merged doc, collection/subject format (e.g. Projects/Oxide) — only reuse an old title if it\'s genuinely the best fit', required: true},
|
|
|
|
|
description: {type: 'string', description: 'One factual sentence describing the merged document\'s subject matter', required: true},
|
|
|
|
|
content: {type: 'string', description: 'Fully reconciled body in markdown, without frontmatter', required: true},
|
|
|
|
|
},
|
|
|
|
|
system: `You are a knowledge base editor merging two overlapping Obsidian documents into one.
|
|
|
|
|
|
|
|
|
|
Structure loosely:
|
|
|
|
|
\`\`\`markdown
|
|
|
|
|
${GENERIC_TEMPLATE}
|
|
|
|
|
\`\`\`
|
|
|
|
|
|
|
|
|
|
Combine both documents, resolve duplication. On contradictions, bias toward whichever document was modified more recently; drop the outdated statement unless the old context adds meaningful nuance.
|
|
|
|
|
|
|
|
|
|
Document A ("${a.name}", last modified ${modifiedOf(a)}):
|
|
|
|
|
\`\`\`markdown
|
|
|
|
|
${stripHeader(a.content)}
|
|
|
|
|
\`\`\`
|
|
|
|
|
|
|
|
|
|
Document B ("${b.name}", last modified ${modifiedOf(b)}):
|
|
|
|
|
\`\`\`markdown
|
|
|
|
|
${stripHeader(b.content)}
|
|
|
|
|
\`\`\``,
|
|
|
|
|
});
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private parseFrontmatter(content: string): {fm: Map<string, string>, body: string} {
|
|
|
|
|
@@ -419,21 +577,30 @@ ${currentBody}
|
|
|
|
|
for (const line of match[1].split('\n')) {
|
|
|
|
|
const i = line.indexOf(':');
|
|
|
|
|
if (i === -1) continue;
|
|
|
|
|
fm.set(line.slice(0, i).trim(), line.slice(i + 1).trim());
|
|
|
|
|
const key = line.slice(0, i).trim();
|
|
|
|
|
const raw = line.slice(i + 1).trim();
|
|
|
|
|
let value = raw;
|
|
|
|
|
try { value = JSON.parse(raw); } catch { /* legacy unquoted value, keep raw */ }
|
|
|
|
|
fm.set(key, value);
|
|
|
|
|
}
|
|
|
|
|
return {fm, body: match[2]};
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Writes the code-owned frontmatter block. `body` is passed through stripHeader() first so a
|
|
|
|
|
* model that ignores instructions and hallucinates its own `---` block can never corrupt or
|
|
|
|
|
* duplicate the real frontmatter — the LLM only ever gets to influence the body.
|
|
|
|
|
*/
|
|
|
|
|
private touchHeader(node: Memory, body: string): string {
|
|
|
|
|
const {fm} = this.parseFrontmatter(node.content);
|
|
|
|
|
fm.set('name', node.name);
|
|
|
|
|
fm.set('description', node.description || '');
|
|
|
|
|
fm.set('modified', new Date().toISOString());
|
|
|
|
|
return this.writeFrontmatter(fm, body);
|
|
|
|
|
return this.writeFrontmatter(fm, stripHeader(body));
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
private writeFrontmatter(fm: Map<string, string>, body: string): string {
|
|
|
|
|
const lines = [...fm.entries()].map(([k, v]) => `${k}: ${v}`);
|
|
|
|
|
const lines = [...fm.entries()].map(([k, v]) => `${k}: ${JSON.stringify(String(v).replace(/\s+/g, ' ').trim())}`);
|
|
|
|
|
return `---\n${lines.join('\n')}\n---\n\n${body.trimStart()}`;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
@@ -452,6 +619,19 @@ ${currentBody}
|
|
|
|
|
return this.access(memories).forget(name);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Ranks a candidate pool by weighted title/description/body similarity against the query embedding */
|
|
|
|
|
private rankByFields(query: number[], candidates: Memory[], limit: number): Memory[] {
|
|
|
|
|
const scored = candidates.map(m => {
|
|
|
|
|
const titleSim = m.titleEmbedding?.length ? 1 - cosineDistance(query, m.titleEmbedding) : 0;
|
|
|
|
|
const descSim = m.embedding?.length ? 1 - cosineDistance(query, m.embedding) : 0;
|
|
|
|
|
const bodySim = m.bodyEmbeddings?.length
|
|
|
|
|
? Math.max(...m.bodyEmbeddings.map(b => 1 - cosineDistance(query, b)))
|
|
|
|
|
: 0;
|
|
|
|
|
return {memory: m, score: titleSim * 0.5 + descSim * 0.35 + bodySim * 0.15};
|
|
|
|
|
});
|
|
|
|
|
return scored.sort((a, b) => b.score - a.score).slice(0, limit).map(s => s.memory);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
async recollect(query: string, memories: Memory[] | MemoryCache, limit = 5, graphDepth = 1): Promise<Memory[]> {
|
|
|
|
|
const store = this.access(memories);
|
|
|
|
|
if (!store.list.length) return [];
|
|
|
|
|
@@ -461,8 +641,11 @@ ${currentBody}
|
|
|
|
|
const [e] = await this.llm.embedding(query);
|
|
|
|
|
if (!e) return [];
|
|
|
|
|
|
|
|
|
|
const vectorResults = store.search(e.embedding, limit);
|
|
|
|
|
const found = new Set<string>(vectorResults.map(r => r.name));
|
|
|
|
|
// Description embedding is the cheap ANN index key; pull a wider pool then re-rank by field weight
|
|
|
|
|
const pool = store.search(e.embedding, Math.max(limit * 3, limit));
|
|
|
|
|
const poolMemories = pool.map(r => store.find(r.name)).filter((m): m is Memory => !!m);
|
|
|
|
|
const ranked = this.rankByFields(e.embedding, poolMemories, limit);
|
|
|
|
|
const found = new Set<string>(ranked.map(m => m.name));
|
|
|
|
|
|
|
|
|
|
if (graphDepth > 0) {
|
|
|
|
|
let frontier = [...found];
|
|
|
|
|
@@ -482,9 +665,9 @@ ${currentBody}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
const vectorOrder = vectorResults.map(r => r.name);
|
|
|
|
|
const graphExpansions = [...found].filter(n => !vectorOrder.includes(n));
|
|
|
|
|
return [...vectorOrder, ...graphExpansions].map(n => store.find(n)!).filter(Boolean);
|
|
|
|
|
const rankedOrder = ranked.map(m => m.name);
|
|
|
|
|
const graphExpansions = [...found].filter(n => !rankedOrder.includes(n));
|
|
|
|
|
return [...rankedOrder, ...graphExpansions].map(n => store.find(n)!).filter(Boolean);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
async memorize(history: LLMMessage[], memories: Memory[] | MemoryCache, options: LLMRequest): Promise<Memory[]> {
|
|
|
|
|
@@ -498,26 +681,40 @@ ${currentBody}
|
|
|
|
|
history.push(pending);
|
|
|
|
|
|
|
|
|
|
const store = this.access(memories);
|
|
|
|
|
const buckets = await this.factAgent(conversation, store, options, this.getWeekMonday());
|
|
|
|
|
const {buckets, journal} = await this.factAgent(conversation, store, options);
|
|
|
|
|
const touched: Memory[] = [];
|
|
|
|
|
|
|
|
|
|
if (journal) {
|
|
|
|
|
const journalName = `Journal/${this.getWeekMonday()}`;
|
|
|
|
|
let jnode = store.find(journalName);
|
|
|
|
|
if (!jnode) {
|
|
|
|
|
jnode = {name: journalName, description: '', content: '', embedding: [], links: [], backlinks: []};
|
|
|
|
|
store.list.push(jnode);
|
|
|
|
|
}
|
|
|
|
|
this.stage(jnode, `### ${new Date().toISOString().slice(0, 10)}\n${journal}`);
|
|
|
|
|
touched.push(jnode);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
for (const {subject, facts} of buckets) {
|
|
|
|
|
let node = store.find(subject);
|
|
|
|
|
const resolved = this.resolveSubject(subject, store);
|
|
|
|
|
let node = store.find(resolved);
|
|
|
|
|
if (!node) {
|
|
|
|
|
node = {name: subject, description: '', content: '', embedding: [], links: [], backlinks: []};
|
|
|
|
|
node = {name: resolved, description: '', content: '', embedding: [], links: [], backlinks: []};
|
|
|
|
|
store.list.push(node);
|
|
|
|
|
}
|
|
|
|
|
this.appendFacts(node, facts);
|
|
|
|
|
const [e] = await this.llm.embedding(node.content);
|
|
|
|
|
if (e) node.embedding = e.embedding;
|
|
|
|
|
this.touch(node.name);
|
|
|
|
|
this.stage(node, facts.map(f => `- ${f}`).join('\n'));
|
|
|
|
|
touched.push(node);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
await Promise.all(touched.map(async node => {
|
|
|
|
|
await embedMemoryFields(node, this.llm);
|
|
|
|
|
this.touch(node.name);
|
|
|
|
|
}));
|
|
|
|
|
|
|
|
|
|
if (touched.length) {
|
|
|
|
|
store.commit();
|
|
|
|
|
store.commit(touched);
|
|
|
|
|
(pending as any).content = `Saved to ${touched.map(n => `[[${n.name}]]`).join(', ')}`;
|
|
|
|
|
await Promise.all(touched.map(node => this.reconcile(node, memories, options).catch(() => {})));
|
|
|
|
|
Promise.all(touched.map(node => this.reconcile(node, memories, options).catch(() => {})));
|
|
|
|
|
} else {
|
|
|
|
|
(pending as any).content = 'Nothing worth remembering.';
|
|
|
|
|
}
|
|
|
|
|
@@ -526,9 +723,9 @@ ${currentBody}
|
|
|
|
|
return touched;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
async reconcileVault(memories: Memory[] | MemoryCache, options: LLMRequest, scope: 'touched' | 'all' = 'touched'): Promise<void> {
|
|
|
|
|
async reconcileAll(memories: Memory[] | MemoryCache, options: LLMRequest, scope: 'touched' | 'all' = 'touched'): Promise<void> {
|
|
|
|
|
const store = this.access(memories);
|
|
|
|
|
const targets = scope === 'all' ? store.list : store.list.filter(m => m.content.includes(FACTS_HEADING));
|
|
|
|
|
const targets = scope === 'all' ? store.list : store.list.filter(m => m.content.includes(PENDING_HEADING));
|
|
|
|
|
await Promise.all(targets.map(node => this.reconcile(node, memories, options)));
|
|
|
|
|
store.commit();
|
|
|
|
|
}
|
|
|
|
|
|