@rohirik/openltm-core 2.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/README.md +67 -0
  2. package/assets/opencode/agents/aegis.md +211 -0
  3. package/assets/opencode/plugins/aegis.ts +3 -0
  4. package/assets/opencode/skills/AgentTrustBoundaries/ContextCrushDefense.md +104 -0
  5. package/assets/opencode/skills/AgentTrustBoundaries/SKILL.md +31 -0
  6. package/assets/opencode/skills/AgentTrustBoundaries/TrustBoundaryPatterns.md +114 -0
  7. package/assets/opencode/skills/AgentTrustBoundaries/Workflows/DefendContextCrush.md +27 -0
  8. package/assets/opencode/skills/AgentTrustBoundaries/Workflows/HandleUntrustedContent.md +27 -0
  9. package/assets/opencode/skills/CommandPathSafety/CommandInjectionPatterns.md +95 -0
  10. package/assets/opencode/skills/CommandPathSafety/PathTraversalAndInstallerSafety.md +106 -0
  11. package/assets/opencode/skills/CommandPathSafety/SKILL.md +31 -0
  12. package/assets/opencode/skills/CommandPathSafety/Workflows/EnforcePathBoundaries.md +27 -0
  13. package/assets/opencode/skills/CommandPathSafety/Workflows/HardenCommandExecution.md +27 -0
  14. package/assets/opencode/skills/SecretSafeHandling/CloudCredentialPatterns.md +106 -0
  15. package/assets/opencode/skills/SecretSafeHandling/SKILL.md +31 -0
  16. package/assets/opencode/skills/SecretSafeHandling/SecretHandlingPlaybook.md +102 -0
  17. package/assets/opencode/skills/SecretSafeHandling/Workflows/DesignSecretSafeFlow.md +27 -0
  18. package/assets/opencode/skills/SecretSafeHandling/Workflows/RemoveSecretExposure.md +27 -0
  19. package/package.json +41 -0
  20. package/src/__tests__/cli/claude.test.ts +122 -0
  21. package/src/__tests__/cli/detect.test.ts +91 -0
  22. package/src/__tests__/cli/install.test.ts +161 -0
  23. package/src/__tests__/cli/opencode.test.ts +169 -0
  24. package/src/__tests__/cli/pi.test.ts +113 -0
  25. package/src/__tests__/cli.test.ts +70 -0
  26. package/src/__tests__/events/crossProcess.test.ts +82 -0
  27. package/src/__tests__/events/index.test.ts +32 -0
  28. package/src/__tests__/extensions.test.ts +81 -0
  29. package/src/__tests__/migrations/retention.test.ts +118 -0
  30. package/src/__tests__/queue/index.test.ts +61 -0
  31. package/src/__tests__/scheduler/index.test.ts +39 -0
  32. package/src/__tests__/vec/index.test.ts +130 -0
  33. package/src/__tests__/vec/parity.test.ts +70 -0
  34. package/src/adapterTypes.ts +23 -0
  35. package/src/cli/_shared.ts +120 -0
  36. package/src/cli/bin.ts +97 -0
  37. package/src/cli/claude.ts +124 -0
  38. package/src/cli/detect.ts +55 -0
  39. package/src/cli/hook.ts +25 -0
  40. package/src/cli/index.ts +22 -0
  41. package/src/cli/install.ts +185 -0
  42. package/src/cli/opencode.ts +193 -0
  43. package/src/cli/pi.ts +74 -0
  44. package/src/cli/types.ts +78 -0
  45. package/src/config.ts +163 -0
  46. package/src/context.ts +172 -0
  47. package/src/dao/conflicts.ts +26 -0
  48. package/src/dao/contextItems.ts +70 -0
  49. package/src/dao/embeddings.ts +78 -0
  50. package/src/dao/index.ts +9 -0
  51. package/src/dao/provenanceAudit.ts +108 -0
  52. package/src/dao/types.ts +142 -0
  53. package/src/db.ts +780 -0
  54. package/src/dedup.ts +12 -0
  55. package/src/embeddings.ts +386 -0
  56. package/src/events/index.ts +130 -0
  57. package/src/extensions.ts +140 -0
  58. package/src/graph.ts +268 -0
  59. package/src/index.ts +95 -0
  60. package/src/janitor/archive.ts +66 -0
  61. package/src/janitor/decay.ts +60 -0
  62. package/src/janitor/dedup.ts +333 -0
  63. package/src/janitor/embeddings.ts +209 -0
  64. package/src/janitor/index.ts +215 -0
  65. package/src/janitor/promote.ts +188 -0
  66. package/src/janitor/providers/anthropic.ts +91 -0
  67. package/src/janitor/providers/cohere.ts +135 -0
  68. package/src/janitor/providers/gemini.ts +156 -0
  69. package/src/janitor/providers/ollama.ts +177 -0
  70. package/src/janitor/providers/openai.ts +121 -0
  71. package/src/janitor/providers/openrouter.ts +182 -0
  72. package/src/janitor/providers/types.ts +154 -0
  73. package/src/janitor/providers/utils.ts +35 -0
  74. package/src/janitor/supersedes.ts +199 -0
  75. package/src/lib/honker.ts +54 -0
  76. package/src/lib/honkerTypes.ts +109 -0
  77. package/src/lib/jsonlLogger.ts +92 -0
  78. package/src/lib/writeQueue.ts +28 -0
  79. package/src/migrations.ts +415 -0
  80. package/src/paths.ts +22 -0
  81. package/src/proposals.ts +120 -0
  82. package/src/providers/disabled.ts +19 -0
  83. package/src/providers/embeddingProvider.ts +49 -0
  84. package/src/providers/gemini.ts +37 -0
  85. package/src/providers/index.ts +2 -0
  86. package/src/providers/ollama.ts +43 -0
  87. package/src/providers/openai.ts +35 -0
  88. package/src/queue/index.ts +53 -0
  89. package/src/queue/worker.ts +77 -0
  90. package/src/recall/categorise.ts +139 -0
  91. package/src/recall/explainer.ts +76 -0
  92. package/src/scheduler/index.ts +97 -0
  93. package/src/schema.sql +191 -0
  94. package/src/secretsScrubber.ts +105 -0
  95. package/src/shared-db.ts +158 -0
  96. package/src/vec/index.ts +161 -0
  97. package/tsconfig.json +9 -0
package/src/dedup.ts ADDED
@@ -0,0 +1,12 @@
1
+ /**
2
+ * Dedup key normalization for memory deduplication.
3
+ * Produces a stable string key from arbitrary content.
4
+ */
5
+ export function normalizeKey(content: string): string {
6
+ return content
7
+ .toLowerCase()
8
+ .replace(/[^\w\s]/g, " ") // strip punctuation
9
+ .replace(/\s+/g, " ") // collapse whitespace
10
+ .trim()
11
+ .substring(0, 200); // cap length
12
+ }
@@ -0,0 +1,386 @@
1
+ /**
2
+ * embeddings.ts — Provider-agnostic embedding utilities for LTM semantic search.
3
+ * Embedding writes/reads now go through src/dao/embeddings.ts (memory_embeddings table).
4
+ * Embedding generation now goes through src/providers/embeddingProvider.ts.
5
+ * Falls back gracefully (returns null) when no provider is configured.
6
+ */
7
+ import type { Database } from "bun:sqlite";
8
+ import type { EmbeddingProvider } from "./providers/embeddingProvider.js";
9
+ import { setEmbedding, listMemoryIdsMissingEmbedding } from "./dao/embeddings.js";
10
+
11
+ // --- Provider config (retained for LLM/auto-relate path) ---
12
+
13
+ type EmbedProvider = "gemini" | "openai" | "openrouter" | "cohere" | "ollama";
14
+
15
+ interface ProviderConfig {
16
+ provider: EmbedProvider;
17
+ apiKey?: string;
18
+ model: string;
19
+ baseUrl?: string;
20
+ }
21
+
22
+ // Caches — stable within a process lifetime (restart required for config changes)
23
+ let _llmConfigCache: ProviderConfig | null | undefined = undefined;
24
+ let _embeddingProviderCache: Promise<EmbeddingProvider> | null = null;
25
+
26
+ /** Shared loader for both embed and llm provider configs. */
27
+ function loadConfig(type: "embed" | "llm"): ProviderConfig | null {
28
+ try {
29
+ const { getDb } = require("./shared-db.js") as typeof import("./shared-db.js");
30
+ const db = getDb();
31
+ const t = type;
32
+ const KEYS = [
33
+ `ltm.${t}.provider`,
34
+ "ltm.gemini.apiKey", `ltm.gemini.${t}Model`,
35
+ "ltm.openai.apiKey", `ltm.openai.${t}Model`,
36
+ "ltm.openrouter.apiKey", `ltm.openrouter.${t}Model`,
37
+ `ltm.cohere.apiKey`, `ltm.cohere.${t}Model`,
38
+ `ltm.ollama.${t}Model`, "ltm.ollama.baseUrl",
39
+ ];
40
+ const placeholders = KEYS.map(() => "?").join(",");
41
+ const rows = db.query<{ key: string; value: string }, string[]>(
42
+ `SELECT key, value FROM settings WHERE key IN (${placeholders})`
43
+ ).all(...KEYS);
44
+ const s = Object.fromEntries(rows.map(r => [r.key, r.value])) as Record<string, string | undefined>;
45
+
46
+ const envProvider = t === "embed" ? process.env.LTM_EMBED_PROVIDER : process.env.LTM_LLM_PROVIDER;
47
+ const provider = (envProvider ?? s[`ltm.${t}.provider`] ?? "gemini") as EmbedProvider;
48
+
49
+ const DEFAULTS: Record<EmbedProvider, { model: string }> = {
50
+ gemini: { model: t === "embed" ? "gemini-embedding-2-preview" : "gemini-2.0-flash-lite" },
51
+ openai: { model: t === "embed" ? "text-embedding-3-small" : "gpt-4o-mini" },
52
+ openrouter: { model: t === "embed" ? "openai/text-embedding-3-large" : "google/gemini-2.0-flash-001" },
53
+ cohere: { model: t === "embed" ? "embed-v4.0" : "command-r-plus" },
54
+ ollama: { model: t === "embed" ? "nomic-embed-text" : "llama3.2" },
55
+ };
56
+
57
+ switch (provider) {
58
+ case "gemini":
59
+ return { provider, apiKey: process.env.GEMINI_API_KEY ?? s["ltm.gemini.apiKey"], model: s[`ltm.gemini.${t}Model`] ?? DEFAULTS.gemini.model };
60
+ case "openai":
61
+ return { provider, apiKey: process.env.OPENAI_API_KEY ?? s["ltm.openai.apiKey"], model: s[`ltm.openai.${t}Model`] ?? DEFAULTS.openai.model };
62
+ case "openrouter":
63
+ return { provider, apiKey: process.env.OPENROUTER_API_KEY ?? s["ltm.openrouter.apiKey"], model: s[`ltm.openrouter.${t}Model`] ?? DEFAULTS.openrouter.model, baseUrl: "https://openrouter.ai/api/v1" };
64
+ case "cohere":
65
+ return { provider, apiKey: process.env.COHERE_API_KEY ?? s["ltm.cohere.apiKey"], model: s[`ltm.cohere.${t}Model`] ?? DEFAULTS.cohere.model };
66
+ case "ollama":
67
+ return { provider, model: s[`ltm.ollama.${t}Model`] ?? DEFAULTS.ollama.model, baseUrl: s["ltm.ollama.baseUrl"] ?? "http://localhost:11434" };
68
+ default:
69
+ return null;
70
+ }
71
+ } catch {
72
+ return null;
73
+ }
74
+ }
75
+
76
+ /** Returns the embedding provider, lazily initialised once per process. */
77
+ async function getEmbeddingProvider(): Promise<EmbeddingProvider> {
78
+ if (!_embeddingProviderCache) {
79
+ _embeddingProviderCache = (async () => {
80
+ const [{ loadConfig }, { loadProvider }] = await Promise.all([
81
+ import("./config.js"),
82
+ import("./providers/embeddingProvider.js"),
83
+ ]);
84
+ const cfg = await loadConfig();
85
+ return loadProvider(cfg.embeddings);
86
+ })();
87
+ }
88
+ return _embeddingProviderCache;
89
+ }
90
+
91
+ export function getLlmConfig(): ProviderConfig | null {
92
+ if (_llmConfigCache !== undefined) return _llmConfigCache;
93
+ _llmConfigCache = loadConfig("llm");
94
+ return _llmConfigCache;
95
+ }
96
+
97
+ // --- Math utils ---
98
+
99
+ export function cosineSimilarity(a: Float32Array, b: Float32Array): number {
100
+ if (a.length !== b.length) return 0;
101
+ let dot = 0, normA = 0, normB = 0;
102
+ for (let i = 0; i < a.length; i++) {
103
+ const ai = a[i] as number;
104
+ const bi = b[i] as number;
105
+ dot += ai * bi;
106
+ normA += ai * ai;
107
+ normB += bi * bi;
108
+ }
109
+ const denom = Math.sqrt(normA) * Math.sqrt(normB);
110
+ return denom === 0 ? 0 : dot / denom;
111
+ }
112
+
113
+ export function vecToBlob(v: Float32Array): Buffer {
114
+ return Buffer.from(v.buffer);
115
+ }
116
+
117
+ export function blobToVec(b: Buffer): Float32Array {
118
+ return new Float32Array(b.buffer, b.byteOffset, b.byteLength / 4);
119
+ }
120
+
121
+ // --- Provider-specific embed implementations ---
122
+
123
+ // Cached Gemini client + the key it was initialized with
124
+ interface GeminiGenerativeModel {
125
+ embedContent(text: string): Promise<{ embedding: { values: number[] } }>;
126
+ }
127
+ interface GeminiAIClient {
128
+ getGenerativeModel(params: { model: string }): GeminiGenerativeModel;
129
+ }
130
+ let _genAI: GeminiAIClient | null = null;
131
+ let _genAIKey: string | undefined;
132
+
133
+ async function embedGemini(text: string, cfg: ProviderConfig): Promise<Float32Array | null> {
134
+ if (!cfg.apiKey) return null;
135
+ if (!_genAI || _genAIKey !== cfg.apiKey) {
136
+ // Dynamic import to avoid compile-time dependency on optional package
137
+ const genAiModule = await (Function('return import("@google/generative-ai")')() as Promise<{
138
+ GoogleGenerativeAI: new (apiKey: string) => GeminiAIClient;
139
+ }>);
140
+ const { GoogleGenerativeAI } = genAiModule;
141
+ _genAI = new GoogleGenerativeAI(cfg.apiKey);
142
+ _genAIKey = cfg.apiKey;
143
+ }
144
+ const model = _genAI.getGenerativeModel({ model: cfg.model });
145
+ const result = await model.embedContent(text);
146
+ return new Float32Array(result.embedding.values);
147
+ }
148
+
149
+ async function embedOpenAICompat(text: string, cfg: ProviderConfig): Promise<Float32Array | null> {
150
+ if (!cfg.apiKey) return null;
151
+ const baseUrl = cfg.baseUrl ?? "https://api.openai.com/v1";
152
+ const res = await fetch(`${baseUrl}/embeddings`, {
153
+ method: "POST",
154
+ headers: { "Authorization": `Bearer ${cfg.apiKey}`, "Content-Type": "application/json" },
155
+ body: JSON.stringify({ model: cfg.model, input: text }),
156
+ });
157
+ if (!res.ok) throw new Error(`${cfg.provider} API error: ${res.status} ${await res.text()}`);
158
+ const json = await res.json() as { data: Array<{ embedding: number[] }> };
159
+ return new Float32Array(json.data[0]!.embedding);
160
+ }
161
+
162
+ async function embedCohere(text: string, cfg: ProviderConfig): Promise<Float32Array | null> {
163
+ if (!cfg.apiKey) return null;
164
+ const res = await fetch("https://api.cohere.com/v2/embed", {
165
+ method: "POST",
166
+ headers: { "Authorization": `Bearer ${cfg.apiKey}`, "Content-Type": "application/json" },
167
+ body: JSON.stringify({ model: cfg.model, texts: [text], input_type: "search_document", embedding_types: ["float"] }),
168
+ });
169
+ if (!res.ok) throw new Error(`cohere API error: ${res.status} ${await res.text()}`);
170
+ const json = await res.json() as { embeddings: { float: number[][] } };
171
+ return new Float32Array(json.embeddings.float[0]!);
172
+ }
173
+
174
+ async function embedOllama(text: string, cfg: ProviderConfig): Promise<Float32Array | null> {
175
+ const res = await fetch(`${cfg.baseUrl}/api/embed`, {
176
+ method: "POST",
177
+ headers: { "Content-Type": "application/json" },
178
+ body: JSON.stringify({ model: cfg.model, input: text }),
179
+ });
180
+ if (!res.ok) throw new Error(`ollama API error: ${res.status} ${await res.text()}`);
181
+ const json = await res.json() as { embeddings: number[][] };
182
+ return new Float32Array(json.embeddings[0]!);
183
+ }
184
+
185
+ // --- Public API ---
186
+
187
+ /**
188
+ * Embed text using the configured provider (from config.json embeddings block).
189
+ * Returns null when provider is disabled or the API call fails.
190
+ */
191
+ export async function embedText(text: string): Promise<Float32Array | null> {
192
+ try {
193
+ const provider = await getEmbeddingProvider();
194
+ if (!await provider.available()) return null;
195
+ return provider.generate(text);
196
+ } catch (e) {
197
+ process.stderr.write(`[embeddings] embedText error: ${e}\n`);
198
+ return null;
199
+ }
200
+ }
201
+
202
+ /**
203
+ * Embed a memory by ID and write the embedding to the memory_embeddings table.
204
+ */
205
+ export async function embedMemory(db: Database, id: number): Promise<void> {
206
+ const row = db.query<{ content: string }, [number]>(
207
+ `SELECT content FROM memories WHERE id=?`
208
+ ).get(id);
209
+ if (!row) return;
210
+
211
+ const provider = await getEmbeddingProvider();
212
+ if (!await provider.available()) return;
213
+
214
+ const vec = await provider.generate(row.content);
215
+ if (!vec) return;
216
+
217
+ await setEmbedding(db, id, vecToBlob(vec), provider.model, provider.dim);
218
+ }
219
+
220
+ /**
221
+ * Back-fill: embed all active memories that have no row in memory_embeddings.
222
+ */
223
+ export async function backfill(db: Database): Promise<void> {
224
+ const provider = await getEmbeddingProvider();
225
+ if (!await provider.available()) {
226
+ process.stderr.write(`[embeddings] Back-fill skipped: provider '${provider.name}' not available\n`);
227
+ return;
228
+ }
229
+
230
+ const ids = listMemoryIdsMissingEmbedding(db, 1000);
231
+ process.stderr.write(`[embeddings] Back-filling ${ids.length} memories...\n`);
232
+
233
+ const BATCH = 20;
234
+ let done = 0;
235
+ for (let i = 0; i < ids.length; i += BATCH) {
236
+ const batch = ids.slice(i, i + BATCH);
237
+ await Promise.all(batch.map(async id => {
238
+ const row = db.query<{ content: string }, [number]>(
239
+ `SELECT content FROM memories WHERE id=?`
240
+ ).get(id);
241
+ if (!row) return;
242
+ const vec = await provider.generate(row.content);
243
+ if (vec) {
244
+ await setEmbedding(db, id, vecToBlob(vec), provider.model, provider.dim);
245
+ done++;
246
+ }
247
+ }));
248
+ process.stderr.write(`[embeddings] ${Math.min(i + BATCH, ids.length)}/${ids.length} done\n`);
249
+ if (i + BATCH < ids.length) await Bun.sleep(200);
250
+ }
251
+ process.stderr.write(`[embeddings] Back-fill complete: ${done}/${ids.length} embedded\n`);
252
+ }
253
+
254
+ // --- Semantic similarity search ---
255
+
256
+ export type SimilarMemory = { id: number; content: string; similarity: number };
257
+
258
+ /**
259
+ * Find the top-N most similar memories to the given text using stored embeddings.
260
+ */
261
+ export async function getSimilarMemories(text: string, topN = 5, threshold = 0.5): Promise<SimilarMemory[]> {
262
+ const vec = await embedText(text);
263
+ if (!vec) return [];
264
+
265
+ const { getDb } = await import("./shared-db.js");
266
+ const db = getDb();
267
+
268
+ // Fast path: sqlite-vec vec0 KNN. Over-fetch so post-filtering on status and
269
+ // threshold still yields topN. Falls through to brute force when the index is
270
+ // unavailable or empty (e.g. a DB that predates the vec backfill).
271
+ const { getCapabilities } = await import("./extensions.js");
272
+ if (getCapabilities().vec) {
273
+ const { knnVec } = await import("./vec/index.js");
274
+ const hits = knnVec(db, vecToBlob(vec), Math.max(topN * 4, topN + 10));
275
+ if (hits.length > 0) {
276
+ const byId = new Map(hits.map(h => [h.id, h.similarity]));
277
+ const ids = hits.map(h => h.id);
278
+ const placeholders = ids.map(() => "?").join(",");
279
+ const rows = db.query<{ id: number; content: string }, number[]>(
280
+ `SELECT id, content FROM memories WHERE status='active' AND id IN (${placeholders})`
281
+ ).all(...ids);
282
+ return rows
283
+ .map(row => ({ id: row.id, content: row.content, similarity: byId.get(row.id) ?? 0 }))
284
+ .filter(r => r.similarity >= threshold)
285
+ .sort((a, b) => b.similarity - a.similarity)
286
+ .slice(0, topN);
287
+ }
288
+ }
289
+
290
+ // Brute-force JS cosine fallback.
291
+ const rows = db.query<{ id: number; content: string; embedding: Buffer }, []>(
292
+ `SELECT m.id, m.content, e.embedding
293
+ FROM memories m JOIN memory_embeddings e ON e.memory_id = m.id
294
+ WHERE m.status = 'active'`
295
+ ).all();
296
+
297
+ return rows
298
+ .map(row => ({ id: row.id, content: row.content, similarity: cosineSimilarity(vec, blobToVec(row.embedding)) }))
299
+ .filter(r => r.similarity >= threshold)
300
+ .sort((a, b) => b.similarity - a.similarity)
301
+ .slice(0, topN);
302
+ }
303
+
304
+ // --- Relation classification ---
305
+
306
+ type AutoRelationType = "supports" | "contradicts" | "refines" | "related_to";
307
+
308
+ const CLASSIFY_PROMPT = `You are a memory relation classifier. Given two facts (A and B), respond with exactly ONE word:
309
+ - "supports" — B reinforces or agrees with A
310
+ - "contradicts" — B conflicts with or contradicts A
311
+ - "refines" — B adds detail or nuance to A
312
+ - "related_to" — B is on the same topic but no clear support/conflict/refinement
313
+ - "none" — B has no meaningful relation to A
314
+
315
+ Respond with exactly one of: supports, contradicts, refines, related_to, none`;
316
+
317
+ const VALID_RELATIONS = new Set<string>(["supports", "contradicts", "refines", "related_to"]);
318
+
319
+ export async function callLlm(
320
+ cfg: ProviderConfig,
321
+ userMessage: string,
322
+ options?: { systemPrompt?: string; maxTokens?: number; raw?: boolean },
323
+ ): Promise<string | null> {
324
+ const body = {
325
+ model: cfg.model,
326
+ messages: [
327
+ { role: "system", content: options?.systemPrompt ?? CLASSIFY_PROMPT },
328
+ { role: "user", content: userMessage },
329
+ ],
330
+ max_tokens: options?.maxTokens ?? 10,
331
+ temperature: 0,
332
+ };
333
+
334
+ try {
335
+ let url: string;
336
+ const headers: Record<string, string> = { "Content-Type": "application/json" };
337
+
338
+ if (cfg.provider === "gemini") {
339
+ url = `https://generativelanguage.googleapis.com/v1beta/openai/chat/completions`;
340
+ headers["Authorization"] = `Bearer ${cfg.apiKey ?? ""}`;
341
+ } else if (cfg.provider === "ollama") {
342
+ url = `${cfg.baseUrl}/v1/chat/completions`;
343
+ } else {
344
+ // openai / openrouter
345
+ url = cfg.baseUrl ? `${cfg.baseUrl}/chat/completions` : "https://api.openai.com/v1/chat/completions";
346
+ headers["Authorization"] = `Bearer ${cfg.apiKey ?? ""}`;
347
+ }
348
+
349
+ const res = await fetch(url, { method: "POST", headers, body: JSON.stringify(body) });
350
+ if (!res.ok) return null;
351
+ const json = await res.json() as { choices?: Array<{ message?: { content?: string } }> };
352
+ const content = json.choices?.[0]?.message?.content ?? null;
353
+ if (content === null) return null;
354
+ return options?.raw ? content.trim() : content.trim().toLowerCase();
355
+ } catch {
356
+ return null;
357
+ }
358
+ }
359
+
360
+ /**
361
+ * Classify the relation between two memory strings using a lightweight LLM call.
362
+ * Returns null if no LLM provider configured or call fails.
363
+ */
364
+ export async function classifyRelation(a: string, b: string): Promise<AutoRelationType | null> {
365
+ const cfg = getLlmConfig();
366
+ if (!cfg) return null;
367
+
368
+ const userMessage = `Memory A: ${a}\n\nMemory B: ${b}`;
369
+ const raw = await callLlm(cfg, userMessage);
370
+ if (!raw || !VALID_RELATIONS.has(raw)) return null;
371
+ return raw as AutoRelationType;
372
+ }
373
+
374
+ // CLI: bun embeddings.ts --backfill
375
+ if (import.meta.main) {
376
+ void (async () => {
377
+ const args = process.argv.slice(2);
378
+ if (args.includes("--backfill")) {
379
+ const { getDb } = await import("./shared-db.js");
380
+ const db = getDb();
381
+ await backfill(db);
382
+ } else {
383
+ process.stderr.write("Usage: bun embeddings.ts --backfill\n");
384
+ }
385
+ })();
386
+ }
@@ -0,0 +1,130 @@
1
+ /**
2
+ * events/index.ts — Honker pub/sub for graph-app liveness.
3
+ *
4
+ * Long-lived processes (the graph server) call `startLtmListener()` once; the
5
+ * loop drains `listen("ltm")` and hands each notification payload to the
6
+ * supplied callback (which drives the existing WebSocket `broadcast()`). Any
7
+ * writer — hooks, the janitor, the worker — calls `notifyLtm()` after a commit
8
+ * to push a liveness event to every connected listener with no polling.
9
+ *
10
+ * Dormant + inert when Honker is unavailable: `notifyLtm()` returns false and
11
+ * `startLtmListener()` returns a non-running handle, so the caller keeps its
12
+ * existing fs.watch + 3s-debounce file-watcher fallback.
13
+ */
14
+ import { getHonker } from "../lib/honker.js";
15
+ import type { HonkerTransaction } from "../lib/honkerTypes.js";
16
+ import { getSetting } from "../shared-db.js";
17
+ import { SETTING_KEYS } from "../janitor/providers/types.js";
18
+
19
+ export const LTM_CHANNEL = "ltm";
20
+
21
+ /** Liveness event `type` for a memory created in another agent process. */
22
+ export const MEMORY_ADDED = "memory_added";
23
+
24
+ /** A liveness event pushed to graph-app — `type` mirrors the WS message kind. */
25
+ export interface LtmLiveEvent {
26
+ readonly type: string;
27
+ readonly [key: string]: unknown;
28
+ }
29
+
30
+ export interface LtmListenerHandle {
31
+ /** Whether the listen loop is running (false without Honker). */
32
+ readonly running: boolean;
33
+ /** Stop the loop and wait for it to settle. */
34
+ stop(): Promise<void>;
35
+ }
36
+
37
+ const INERT: LtmListenerHandle = { running: false, stop: async () => {} };
38
+
39
+ /**
40
+ * Push a liveness event on the "ltm" channel. Returns false when Honker is
41
+ * unavailable (caller's file-watcher fallback still fires). Best-effort: a
42
+ * notify failure never throws.
43
+ */
44
+ export function notifyLtm(event: LtmLiveEvent, opts?: { tx?: HonkerTransaction }): boolean {
45
+ const h = getHonker();
46
+ if (!h) return false;
47
+ try {
48
+ h.notify(LTM_CHANNEL, event, opts?.tx ? { tx: opts.tx } : undefined);
49
+ return true;
50
+ } catch {
51
+ return false;
52
+ }
53
+ }
54
+
55
+ /**
56
+ * Start the "ltm" channel listen loop. Each notification's payload is passed to
57
+ * `onEvent`. No-op (inert handle) when Honker is unavailable.
58
+ */
59
+ export function startLtmListener(
60
+ onEvent: (event: LtmLiveEvent) => void,
61
+ opts?: { pollMs?: number },
62
+ ): LtmListenerHandle {
63
+ const h = getHonker();
64
+ if (!h) return INERT;
65
+
66
+ const controller = new AbortController();
67
+ const loop = (async () => {
68
+ const listenOpts = { signal: controller.signal, ...(opts?.pollMs != null ? { pollMs: opts.pollMs } : {}) };
69
+ for await (const note of h.listen(LTM_CHANNEL, listenOpts)) {
70
+ if (controller.signal.aborted) return;
71
+ const payload = note.payload;
72
+ if (payload && typeof payload === "object" && "type" in payload) {
73
+ try { onEvent(payload as LtmLiveEvent); } catch { /* dispatch best-effort */ }
74
+ }
75
+ }
76
+ })();
77
+ loop.catch(() => {});
78
+
79
+ return {
80
+ running: true,
81
+ stop: async () => {
82
+ controller.abort();
83
+ await loop.catch(() => {});
84
+ },
85
+ };
86
+ }
87
+
88
+ /**
89
+ * Whether opt-in cross-process memory sync is enabled (setting
90
+ * `ltm.crossProcessSync`, default "off"). Cross-process push additionally
91
+ * requires a live Honker handle — this only reports the user's intent.
92
+ */
93
+ export function isCrossProcessSyncEnabled(): boolean {
94
+ return getSetting(SETTING_KEYS.CROSS_PROCESS_SYNC) === "on";
95
+ }
96
+
97
+ /**
98
+ * Push a `memory_added` event so sibling agent processes (Claude Code,
99
+ * OpenCode, Pi) sharing one openltm.db can react to a memory created elsewhere.
100
+ * No-op unless the `ltm.crossProcessSync` flag is on AND Honker is available.
101
+ * Returns true only when the notify was actually sent.
102
+ */
103
+ export function notifyMemoryAdded(
104
+ memory: { id: number; project_scope?: string | null },
105
+ opts?: { tx?: HonkerTransaction },
106
+ ): boolean {
107
+ if (!isCrossProcessSyncEnabled()) return false;
108
+ return notifyLtm(
109
+ { type: MEMORY_ADDED, id: memory.id, project_scope: memory.project_scope ?? null, pid: process.pid },
110
+ opts,
111
+ );
112
+ }
113
+
114
+ /**
115
+ * Adapter-facing listener: invoke `onMemoryAdded` for each `memory_added` event
116
+ * raised by another process (ignores this process's own events by pid). Inert
117
+ * (non-running handle) unless the flag is on AND Honker is available, so an
118
+ * adapter can call this unconditionally and degrade to a no-op.
119
+ */
120
+ export function startCrossProcessSync(
121
+ onMemoryAdded: (event: LtmLiveEvent) => void,
122
+ opts?: { pollMs?: number },
123
+ ): LtmListenerHandle {
124
+ if (!isCrossProcessSyncEnabled()) return INERT;
125
+ return startLtmListener(event => {
126
+ if (event.type !== MEMORY_ADDED) return;
127
+ if (event["pid"] === process.pid) return; // drop our own echo
128
+ onMemoryAdded(event);
129
+ }, opts);
130
+ }
@@ -0,0 +1,140 @@
1
+ /**
2
+ * extensions.ts — Loadable SQLite extension capability probe + activation.
3
+ *
4
+ * Bun's bundled SQLite is compiled WITHOUT dynamic extension loading, so
5
+ * db.loadExtension() throws "does not support dynamic extension loading".
6
+ * We switch the process to a system extension-enabled libsqlite3 via
7
+ * Database.setCustomSQLite() (a process-global static that MUST run before the
8
+ * first Database is opened), then loadExtension() per-connection.
9
+ *
10
+ * Everything degrades gracefully: any failure (no system sqlite, missing
11
+ * binary, force-disabled via env) leaves the capability false and callers
12
+ * fall back to the pure-JS / FTS-only paths. This module NEVER throws.
13
+ */
14
+ import { Database } from "bun:sqlite";
15
+ import { existsSync } from "fs";
16
+
17
+ export interface Capabilities {
18
+ /** A system extension-enabled SQLite is active for this process. */
19
+ customSqlite: boolean;
20
+ /** sqlite-vec (vec0 virtual tables / KNN) is loaded. */
21
+ vec: boolean;
22
+ /** Honker (durable queue / scheduler / pub-sub) extension is loaded. */
23
+ honker: boolean;
24
+ }
25
+
26
+ const CAPS_NONE: Capabilities = { customSqlite: false, vec: false, honker: false };
27
+
28
+ let _caps: Capabilities | null = null;
29
+ let _customSqliteApplied = false;
30
+
31
+ // Probe order: explicit override first, then common Homebrew / Linux locations.
32
+ function systemSqliteCandidates(): string[] {
33
+ const out: string[] = [];
34
+ const override = process.env["LTM_SQLITE_LIB"];
35
+ if (override) out.push(override);
36
+ out.push(
37
+ "/opt/homebrew/opt/sqlite/lib/libsqlite3.dylib",
38
+ "/usr/local/opt/sqlite/lib/libsqlite3.dylib",
39
+ "/usr/lib/x86_64-linux-gnu/libsqlite3.so.0",
40
+ "/usr/lib/aarch64-linux-gnu/libsqlite3.so.0",
41
+ "/usr/lib64/libsqlite3.so.0",
42
+ );
43
+ return out;
44
+ }
45
+
46
+ function envDisabled(name: string): boolean {
47
+ const v = process.env[name];
48
+ return v === "1" || v === "true";
49
+ }
50
+
51
+ /** Locate a system extension-enabled libsqlite3, or null if none found. */
52
+ export function locateSystemSqlite(): string | null {
53
+ for (const p of systemSqliteCandidates()) {
54
+ try {
55
+ if (existsSync(p)) return p;
56
+ } catch {
57
+ /* ignore unreadable candidate */
58
+ }
59
+ }
60
+ return null;
61
+ }
62
+
63
+ /**
64
+ * Switch the process to a system extension-enabled SQLite. Idempotent and
65
+ * never throws. Must run before the first Database is constructed to take
66
+ * effect. Returns true if a custom sqlite is (now) active.
67
+ *
68
+ * Skipped entirely when BOTH vec and honker are force-disabled — there is no
69
+ * reason to leave Bun's bundled SQLite in that case.
70
+ */
71
+ export function ensureCustomSqlite(): boolean {
72
+ if (_customSqliteApplied) return true;
73
+ if (envDisabled("LTM_DISABLE_VEC") && envDisabled("LTM_DISABLE_HONKER")) return false;
74
+ const lib = locateSystemSqlite();
75
+ if (!lib) return false;
76
+ try {
77
+ (Database as unknown as { setCustomSQLite: (p: string) => void }).setCustomSQLite(lib);
78
+ _customSqliteApplied = true;
79
+ return true;
80
+ } catch {
81
+ return false;
82
+ }
83
+ }
84
+
85
+ function loadVec(db: Database): boolean {
86
+ try {
87
+ // Resolve the loadable path synchronously and load it directly — avoids
88
+ // any async-import concern in synchronous DB-init paths.
89
+ // eslint-disable-next-line @typescript-eslint/no-var-requires
90
+ const mod = require("sqlite-vec") as { getLoadablePath?: () => string };
91
+ const path = mod.getLoadablePath?.();
92
+ if (!path) return false;
93
+ db.loadExtension(path);
94
+ db.query("SELECT vec_version()").get();
95
+ return true;
96
+ } catch {
97
+ return false;
98
+ }
99
+ }
100
+
101
+ function loadHonker(db: Database): boolean {
102
+ const ext = process.env["LTM_HONKER_EXT"];
103
+ if (!ext || !existsSync(ext)) return false;
104
+ try {
105
+ db.loadExtension(ext);
106
+ return true;
107
+ } catch {
108
+ return false;
109
+ }
110
+ }
111
+
112
+ /**
113
+ * Load available extensions into the given connection and cache the resulting
114
+ * capabilities for the process. Never throws.
115
+ *
116
+ * @param opts.vec attempt sqlite-vec (default true; env LTM_DISABLE_VEC wins)
117
+ * @param opts.honker attempt Honker (default false; env LTM_DISABLE_HONKER wins,
118
+ * and a libhonker_ext path must be set via LTM_HONKER_EXT)
119
+ */
120
+ export function loadExtensions(db: Database, opts?: { vec?: boolean; honker?: boolean }): Capabilities {
121
+ const wantVec = (opts?.vec ?? true) && !envDisabled("LTM_DISABLE_VEC");
122
+ const wantHonker = (opts?.honker ?? false) && !envDisabled("LTM_DISABLE_HONKER");
123
+
124
+ const customSqlite = ensureCustomSqlite();
125
+ const vec = customSqlite && wantVec ? loadVec(db) : false;
126
+ const honker = customSqlite && wantHonker ? loadHonker(db) : false;
127
+
128
+ _caps = { customSqlite, vec, honker };
129
+ return _caps;
130
+ }
131
+
132
+ /** Return the cached capabilities, or all-false if loadExtensions hasn't run. */
133
+ export function getCapabilities(): Capabilities {
134
+ return _caps ?? { ...CAPS_NONE };
135
+ }
136
+
137
+ /** Test-only: clear cached capability state (does NOT undo setCustomSQLite). */
138
+ export function resetCapabilitiesForTesting(): void {
139
+ _caps = null;
140
+ }