cogmemory 0.0.1-beta.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,878 @@
1
+ import type {
2
+ ArbiterEvaluationResult,
3
+ MemoryInclusionReason,
4
+ MemoryInjectionEntry,
5
+ MemoryInjectionReport,
6
+ MemoryReconciliation,
7
+ ArbiterFn,
8
+ CognitiveMemoryOptions,
9
+ CognitiveMemoryStateSnapshot,
10
+ KnowledgeTension,
11
+ MemoryItem,
12
+ MemoryTier,
13
+ ProprioceptiveSelfModel,
14
+ } from "./types.js";
15
+ import { runFastGate } from "./fast-gate.js";
16
+ import {
17
+ CANDIDATE_FLOOR,
18
+ COLLAPSE_THRESHOLD,
19
+ estimateTokens,
20
+ gistOf,
21
+ distinctiveTokens,
22
+ isInteractionScoped,
23
+ isLossyRewrite,
24
+ MAX_CANDIDATES,
25
+ overlapScore,
26
+ PROMOTE_THRESHOLD,
27
+ relevanceTokens,
28
+ similarity
29
+ } from "./relevance.js";
30
+
31
+ /**
32
+ * CognitiveMemory
33
+ *
34
+ * An intelligent 4-tier cache layer (L0-L3) with:
35
+ * - Anticipatory pre-staging (predicts next turn's needs)
36
+ * - Tension tracking (pinned contradictions in L0)
37
+ * - Proprioceptive self-model (guards weak domains)
38
+ * - Zero added TTFT latency (runs asynchronously post-turn)
39
+ */
40
+ /**
41
+ * Relevance scoring, shared by automatic promotion and on-demand `search()`.
42
+ *
43
+ * It used to be a single domain string compared with `turn.includes(domain)`,
44
+ * which almost never matched: a memory tagged "naming-conventions" cannot match
45
+ * a question that says "naming". Overlap over content words works because the
46
+ * memory text and the question share the words that identify the fact.
47
+ */
48
+ /*
49
+ * Relevance scoring, merge safety and index formatting all live in
50
+ * `./relevance.ts` rather than here.
51
+ *
52
+ * They used to be private copies in this file, and the same logic was separately
53
+ * reimplemented in the service. Three copies of a merge rule is three chances to
54
+ * disagree about whether a fact was lost, so there is now exactly one, and it is
55
+ * the version with the hostname and containment fixes in it.
56
+ */
57
+
58
+ /** L1 only starts evicting past this size. */
59
+ const DEMOTE_ABOVE = 5
60
+
61
+
62
+ export class CognitiveMemory {
63
+ // L0: Pinned core state (identity, self-model, active tensions)
64
+ private activeTensions: Map<string, KnowledgeTension> = new Map();
65
+ private selfModel: ProprioceptiveSelfModel;
66
+ private activeTaskTrace = "";
67
+
68
+ // L1: Hot Cache (pre-staged for upcoming turn)
69
+ private l1HotCache: Map<string, MemoryItem> = new Map();
70
+
71
+ // L2: Warm Storage (indexed candidates ready for promotion)
72
+ private l2WarmStore: Map<string, MemoryItem> = new Map();
73
+
74
+ // L3: Cold Archive (historical, loaded on demand)
75
+ private l3ColdArchive: Map<string, MemoryItem> = new Map();
76
+
77
+ private maxL0Tokens: number;
78
+ private maxL1Tokens: number;
79
+ /** Ceiling on everything injected into one prompt, index and bodies together. */
80
+ private maxTotalTokens: number;
81
+ private arbiter: ArbiterFn | null = null;
82
+ private autoExtractMemories: boolean;
83
+ /** Model-backed turn extractor; the regex fallback runs when this is absent. */
84
+ private reconcile?: CognitiveMemoryOptions["reconcile"];
85
+ private extract?: (turn: { userMessage: string; assistantResponse: string }) => Promise<{
86
+ memories: Array<{ content: string; domains?: string[] }>;
87
+ tensions?: Array<{
88
+ claimA: string;
89
+ claimB: string;
90
+ impact: "low" | "medium" | "critical";
91
+ actionableQuestion: string;
92
+ }>;
93
+ }>;
94
+ private onPersist?: (state: CognitiveMemoryStateSnapshot) => Promise<void> | void;
95
+
96
+ private stats = {
97
+ totalTurnsProcessed: 0,
98
+ predictionsHit: 0,
99
+ predictionsTotal: 0,
100
+ tensionsDetected: 0,
101
+ };
102
+
103
+ constructor(options: CognitiveMemoryOptions = {}) {
104
+ this.maxL0Tokens = options.maxL0Tokens ?? 2000;
105
+ this.maxL1Tokens = options.maxL1Tokens ?? 8000;
106
+ this.maxTotalTokens = options.maxTotalTokens ?? 2000;
107
+ this.arbiter = options.arbiter ?? null;
108
+ this.autoExtractMemories = options.autoExtractMemories ?? true;
109
+ this.extract = options.extract;
110
+ this.reconcile = options.reconcile;
111
+ this.onPersist = options.onPersist;
112
+
113
+ this.selfModel = {
114
+ domains: options.initialSelfModel?.domains ?? {},
115
+ calibrationFactor: options.initialSelfModel?.calibrationFactor ?? 1.0,
116
+ activeDomains: options.initialSelfModel?.activeDomains ?? [],
117
+ };
118
+ }
119
+
120
+ /**
121
+ * Sub-1ms synchronous call to produce the prompt context block.
122
+ * Concatenates L0 (active tensions + self-model) and pre-staged L1.
123
+ * Adds zero latency to TTFT.
124
+ */
125
+ /**
126
+ * Decide what goes into this turn's prompt, and record why.
127
+ *
128
+ * Follows the index/body split that Claude Code's `MEMORY.md` and Letta's
129
+ * progressive disclosure both converged on: a short index of every memory is
130
+ * always present, and a body is only spent where a deterministic signal earned
131
+ * it. Always injecting bodies costs an order of magnitude more tokens and, per
132
+ * Chroma's context-rot work, injects distractors by construction.
133
+ *
134
+ * @param forceFull memory ids whose body should be included regardless of tier.
135
+ */
136
+ planInjection(options: { userMessage?: string; forceFull?: Iterable<string> } = {}): MemoryInjectionReport {
137
+ const force = new Set(options.forceFull ?? []);
138
+ const sections: string[] = [];
139
+ const entries: MemoryInjectionEntry[] = [];
140
+ let used = 0;
141
+ let truncated = false;
142
+
143
+ const include = (
144
+ item: MemoryItem,
145
+ reason: MemoryInclusionReason,
146
+ withBody: boolean
147
+ ): void => {
148
+ const gist = gistOf(item);
149
+ const body = withBody ? item.content : undefined;
150
+ const tags = item.metadata.domains.slice(0, 3);
151
+ const suffix = tags.length > 0 ? ` (${tags.join(", ")})` : "";
152
+ const tokens = estimateTokens(`${gist}${suffix}${body ?? ""}`);
153
+ if (used + tokens > this.maxTotalTokens) {
154
+ truncated = true;
155
+ return;
156
+ }
157
+ used += tokens;
158
+ entries.push({ id: item.id, tier: item.tier, reason, gist, body, tokens });
159
+ };
160
+
161
+ // Fast gate: catch a contradiction in what the user just said.
162
+ if (options.userMessage) {
163
+ const gate = runFastGate(options.userMessage);
164
+ if (gate.action === "inject_caution" && gate.cautionNote) {
165
+ sections.push(`### ⚠️ Correction Detected In This Message\n${gate.cautionNote}`);
166
+ }
167
+ }
168
+
169
+ // Weak-domain warnings: domains this agent is unreliable in.
170
+ const weakDomains = this.selfModel.activeDomains
171
+ .map((d) => ({ d, capability: this.selfModel.domains[d] }))
172
+ .filter((x) => x.capability && x.capability.reliabilityScore < 0.75);
173
+ if (weakDomains.length > 0) {
174
+ const lines = weakDomains.map(({ d, capability }) => {
175
+ const cap = capability!;
176
+ include(
177
+ {
178
+ id: `guardrail-${d}`,
179
+ content: `${d}: reliability ${Math.round(cap.reliabilityScore * 100)}% over ${cap.sampleCount} tasks.${
180
+ cap.knownFailurePatterns.length > 0 ? ` Known pitfalls: ${cap.knownFailurePatterns.join("; ")}` : ""
181
+ }${cap.recommendedStrategies.length > 0 ? ` Approach: ${cap.recommendedStrategies.join("; ")}` : ""}`,
182
+ bookmark: `${d} weak domain`,
183
+ tier: "L0",
184
+ metadata: { domains: [d], createdAt: Date.now(), lastAccessedAt: Date.now(), accessCount: 0 },
185
+ },
186
+ "guardrail",
187
+ true,
188
+ );
189
+ return `- ${d} (${Math.round(cap.reliabilityScore * 100)}% reliable): ${
190
+ cap.knownFailurePatterns.join("; ") || "be careful"
191
+ }`;
192
+ });
193
+ sections.push(`### Weak Domains — Under 75% Reliability\n${lines.join("\n")}`);
194
+ }
195
+
196
+ // Index: one line per memory, bodies withheld unless a trigger earned them.
197
+ const indexLines: string[] = [];
198
+ for (const item of this.l1HotCache.values()) {
199
+ const tags = item.metadata.domains.slice(0, 3);
200
+ const suffix = tags.length > 0 ? ` (${tags.join(", ")})` : "";
201
+ if (force.has(item.id)) {
202
+ include(item, "trigger", true);
203
+ indexLines.push(`- ${item.content}${suffix}`);
204
+ } else {
205
+ include(item, "index", false);
206
+ indexLines.push(`- ${gistOf(item)}${suffix}`);
207
+ }
208
+ }
209
+ if (indexLines.length > 0) {
210
+ sections.push(
211
+ [
212
+ "### Memory index — established earlier in this project",
213
+ "One line per remembered item. Use `recall` to pull a full item, then treat it as true.",
214
+ ...indexLines,
215
+ ].join("\n"),
216
+ );
217
+ }
218
+
219
+ // Unresolved contradictions earn full bodies: they are prompts to clarify, not trivia.
220
+ const activeTensions = [...this.activeTensions.values()].filter((t) => t.status === "active");
221
+ if (activeTensions.length > 0) {
222
+ const lines = activeTensions.map((t) => {
223
+ include(
224
+ {
225
+ id: t.id,
226
+ content: `${t.claimA.statement} vs ${t.claimB.statement} — ${t.actionableQuestion}`,
227
+ bookmark: t.claimA.statement,
228
+ tier: "L0",
229
+ metadata: { domains: [], createdAt: t.claimA.timestamp, lastAccessedAt: Date.now(), accessCount: 0 },
230
+ },
231
+ "tension",
232
+ true,
233
+ );
234
+ return `- [${t.impact.toUpperCase()}] "${t.claimA.statement}" conflicts with "${t.claimB.statement}". Ask: ${t.actionableQuestion}`;
235
+ });
236
+ sections.push(`### Unresolved Contradictions\n${lines.join("\n")}`);
237
+ }
238
+
239
+ return {
240
+ text: sections.length > 0 ? `\n## Cognitive Memory State\n${sections.join("\n\n")}\n` : "",
241
+ entries,
242
+ totalTokens: used,
243
+ truncated,
244
+ };
245
+ }
246
+
247
+ /** Prompt text only. Prefer `planInjection` when you want to log what went in. */
248
+ getPromptContext(currentUserMessage?: string): string {
249
+ return this.planInjection({ userMessage: currentUserMessage }).text;
250
+ }
251
+
252
+ /**
253
+ * Search every tier for memories relevant to `query`.
254
+ *
255
+ * This is the on-demand path, exposed to the agent as a tool. Pre-staging into
256
+ * the prompt is a best-effort optimisation; a model that does not read the
257
+ * block, or whose question shares no words with it, can still ask. Ranking is
258
+ * deterministic — no model involved — so recall does not depend on model
259
+ * quality.
260
+ */
261
+ search(query: string, limit = 8): Array<{ item: MemoryItem; score: number }> {
262
+ const q = relevanceTokens(query);
263
+ if (q.size === 0) return [];
264
+
265
+ const pool: MemoryItem[] = [
266
+ ...this.l1HotCache.values(),
267
+ ...this.l2WarmStore.values(),
268
+ ...this.l3ColdArchive.values(),
269
+ ];
270
+
271
+ const scored = pool
272
+ .map((item) => ({
273
+ item,
274
+ score: overlapScore(q, relevanceTokens(`${item.content} ${item.metadata.domains.join(" ")}`)),
275
+ }))
276
+ .filter((entry) => entry.score > 0)
277
+ .sort((a, b) => b.score - a.score || a.item.content.length - b.item.content.length);
278
+
279
+ // Collapse paraphrases of the same fact to the best-scoring instance.
280
+ const seen: string[] = [];
281
+ const out: Array<{ item: MemoryItem; score: number }> = [];
282
+ for (const entry of scored) {
283
+ // Identical-on-identity-tokens, not identical-on-words: a paraphrase of one
284
+ // fact should occupy one slot in a result list.
285
+ if (seen.some((existing) => similarity(existing, entry.item.content) >= COLLAPSE_THRESHOLD)) {
286
+ continue;
287
+ }
288
+ seen.push(entry.item.content);
289
+ out.push(entry);
290
+ if (out.length >= Math.max(1, limit)) break;
291
+ }
292
+ return out;
293
+ }
294
+
295
+ /**
296
+ * Post-turn asynchronous execution.
297
+ * Fired when the agent finishes streaming its response to the user.
298
+ * Evaluates turn, manages L1 cache, detects tensions, updates self-model.
299
+ */
300
+ async postTurnAsync(params: {
301
+ userMessage: string;
302
+ assistantResponse: string;
303
+ detectedDomains?: string[];
304
+ }): Promise<void> {
305
+ this.stats.totalTurnsProcessed += 1;
306
+
307
+ // 1. Update active domains
308
+ if (params.detectedDomains && params.detectedDomains.length > 0) {
309
+ this.selfModel.activeDomains = Array.from(
310
+ new Set([...this.selfModel.activeDomains, ...params.detectedDomains])
311
+ );
312
+ }
313
+
314
+ // 2. Select candidates from L2 for Arbiter evaluation
315
+ const candidates = Array.from(this.l2WarmStore.values())
316
+ .slice(0, 20)
317
+ .map((item) => ({
318
+ id: item.id,
319
+ bookmark: item.bookmark,
320
+ domains: item.metadata.domains,
321
+ hasTension: false,
322
+ }));
323
+
324
+ const l1Summaries = Array.from(this.l1HotCache.values()).map((m) => ({
325
+ id: m.id,
326
+ bookmark: m.bookmark,
327
+ domains: m.metadata.domains,
328
+ }));
329
+
330
+ // 3. Run Arbiter if configured, or use built-in heuristic arbiter
331
+ const evaluation = this.arbiter
332
+ ? await this.arbiter({
333
+ turnText: params.userMessage,
334
+ assistantReply: params.assistantResponse,
335
+ l0Prompt: this.activeTaskTrace,
336
+ l1Summaries,
337
+ candidates,
338
+ })
339
+ : this.runHeuristicArbiter(params, candidates);
340
+
341
+ // 4. Apply Arbiter Decision
342
+ this.applyArbiterDecision(evaluation);
343
+
344
+ // 5. Learn from the turn. A model-backed extractor is preferred: the regex
345
+ // fallback only catches "always/never/remember to", so plain project
346
+ // facts ("the staging URL is X") were never learned at all.
347
+ if (this.autoExtractMemories) {
348
+ if (this.extract) {
349
+ try {
350
+ await this.applyExtraction(
351
+ await this.extract({ userMessage: params.userMessage, assistantResponse: params.assistantResponse }),
352
+ );
353
+ } catch {
354
+ // A failed extraction must not break the turn; the regex fallback below
355
+ // still gets a chance.
356
+ this.autoExtractTurnMemory(params.userMessage, params.assistantResponse);
357
+ }
358
+ } else if (params.assistantResponse.length > 50) {
359
+ this.autoExtractTurnMemory(params.userMessage, params.assistantResponse);
360
+ }
361
+ }
362
+
363
+ // 6. Enforce the L0/L1 budgets so the "cache" cannot grow without bound.
364
+ this.enforceBudgets();
365
+
366
+ // 7. Trigger persistence callback if registered
367
+ if (this.onPersist) {
368
+ try {
369
+ await this.onPersist(this.getSnapshot());
370
+ } catch {
371
+ // Non-blocking
372
+ }
373
+ }
374
+ }
375
+
376
+ /**
377
+ * Fold extracted memories and tensions in, skipping anything already known.
378
+ *
379
+ * Extracted items land in L2 rather than L1: they are candidates, and the
380
+ * arbiter decides what is worth pre-staging. That is what gives the tiering
381
+ * something to actually do — previously nothing ever wrote to L2, so every
382
+ * arbiter run evaluated an empty candidate set.
383
+ */
384
+ private async applyExtraction(result: {
385
+ memories: Array<{ content: string; domains?: string[] }>;
386
+ tensions?: Array<{
387
+ claimA: string;
388
+ claimB: string;
389
+ impact: "low" | "medium" | "critical";
390
+ actionableQuestion: string;
391
+ }>;
392
+ }): Promise<void> {
393
+ // Memories held aside for one batched adjudication at the end of the turn.
394
+ const pending: Array<{ content: string; candidates: MemoryItem[] }> = [];
395
+ for (const memory of result.memories ?? []) {
396
+ const content = memory.content.trim();
397
+ if (content.length < 8 || content.length > 600) continue;
398
+ if (isInteractionScoped(content)) continue;
399
+
400
+ // Two stages: recall candidates cheaply, then adjudicate.
401
+ //
402
+ // The safe default when nothing can adjudicate, or when the adjudicator is
403
+ // unsure, is to add. An unmerged duplicate costs one row; a wrong merge
404
+ // corrupts what we believe and is hard to unwind. Zep publishes the same
405
+ // bias: prefer under-merge over over-merge.
406
+ let stored = content;
407
+ const candidates = this.recallSimilar(content);
408
+ if (candidates.length > 0) {
409
+ let verdict: MemoryReconciliation = { action: "add" };
410
+ if (this.reconcile) {
411
+ // Deferred: one adjudicator call for the whole turn, not one per
412
+ // memory. Awaiting inside this loop made a five-memory turn cost six
413
+ // serial model calls.
414
+ // Added optimistically below, then the batch verdict may reject,
415
+ // merge or replace it. Storing first keeps memory available even if
416
+ // the adjudicator is slow or never answers.
417
+ pending.push({ content, candidates });
418
+ } else if (candidates.some((c) => c.content.trim().toLowerCase() === content.trim().toLowerCase())) {
419
+ // Byte-identical restatement: never worth storing twice.
420
+ verdict = { action: "merge" };
421
+ }
422
+
423
+ if (verdict.action === "reject") continue;
424
+ if (verdict.action === "merge") {
425
+ // Merge means one memory survives, holding the fuller statement -
426
+ // never "keep both" and never "keep the barer one". A rewrite that
427
+ // would drop information is declined, and falling through stores the
428
+ // restatement separately instead of losing what it added.
429
+ // The incoming statement was never stored, so there is nothing to
430
+ // remove: it folds into the survivor or falls through and is kept.
431
+ if (candidates[0] && this.mergeInto(candidates[0], verdict.content, { text: content, stored: false })) {
432
+ continue;
433
+ }
434
+ }
435
+ if (verdict.action === "replace") {
436
+ // Supersede rather than keep both: the old entry stops being returned.
437
+ this.supersede(candidates.map((c) => c.content));
438
+ }
439
+ stored = verdict.content?.trim() || content;
440
+ }
441
+
442
+ const now = Date.now();
443
+ const id = `mem-${now}-${Math.random().toString(36).slice(2, 7)}`;
444
+ // File it in L1, not L2. The user stated this one turn ago, so it is
445
+ // relevant by definition: waiting for the arbiter to promote it added a
446
+ // one-turn lag, which meant a freshly-taught fact was still missing from
447
+ // the very next prompt.
448
+ this.addMemory(
449
+ {
450
+ id,
451
+ content: stored,
452
+ bookmark: stored.slice(0, 80),
453
+ tier: "L1",
454
+ metadata: {
455
+ domains: memory.domains ?? [],
456
+ createdAt: now,
457
+ lastAccessedAt: now,
458
+ accessCount: 1,
459
+ },
460
+ },
461
+ "L1",
462
+ );
463
+ }
464
+
465
+ // One adjudicator call for the whole turn. Failures fall back to "keep
466
+ // both", so a flaky model costs duplicates, never lost information.
467
+ if (pending.length > 0 && this.reconcile) {
468
+ let verdicts: MemoryReconciliation[] = [];
469
+ try {
470
+ verdicts = await this.reconcile({
471
+ items: pending.map((p) => ({ candidate: p.content, remember: p.candidates.map((c) => c.content) })),
472
+ });
473
+ } catch {
474
+ verdicts = [];
475
+ }
476
+ for (const [index, entry] of pending.entries()) {
477
+ const verdict = verdicts[index];
478
+ if (!verdict) continue;
479
+ if (verdict.action === "reject") this.removeByContent(entry.content);
480
+ else if (verdict.action === "merge") {
481
+ const survivor = entry.candidates[0];
482
+ // Stored optimistically, so the duplicate is removed on the way in -
483
+ // before the rewrite, never after.
484
+ if (survivor && this.mergeInto(survivor, verdict.content, { text: entry.content, stored: true })) {
485
+ continue;
486
+ }
487
+ } else if (verdict.action === "replace") {
488
+ this.supersede(entry.candidates.map((c) => c.content));
489
+ if (verdict.content) this.replaceByContent(entry.content, verdict.content);
490
+ }
491
+ }
492
+ }
493
+
494
+ for (const tension of result.tensions ?? []) {
495
+ const key = `${tension.claimA}::${tension.claimB}`.toLowerCase();
496
+ if (this.activeTensions.has(key)) continue;
497
+ this.addTension({
498
+ id: key,
499
+ status: "active",
500
+ claimA: { source: "user", statement: tension.claimA, timestamp: Date.now() },
501
+ claimB: { source: "conversation", statement: tension.claimB, timestamp: Date.now() },
502
+ impact: tension.impact,
503
+ taskRelevance: 1,
504
+ actionableQuestion: tension.actionableQuestion,
505
+ });
506
+ }
507
+ }
508
+
509
+ /** Drop an entry by its exact content. */
510
+ private removeByContent(content: string): void {
511
+ const target = content.trim().toLowerCase();
512
+ for (const map of [this.l1HotCache, this.l2WarmStore, this.l3ColdArchive]) {
513
+ for (const [id, item] of map) {
514
+ if (item.content.trim().toLowerCase() === target) map.delete(id);
515
+ }
516
+ }
517
+ }
518
+
519
+ /** Rewrite one entry's text in place, matched on its old value. */
520
+ private replaceByContent(from: string, to: string): void {
521
+ const target = from.trim().toLowerCase();
522
+ for (const map of [this.l1HotCache, this.l2WarmStore, this.l3ColdArchive]) {
523
+ for (const item of map.values()) {
524
+ if (item.content.trim().toLowerCase() === target && !isLossyRewrite(item.content, to)) {
525
+ this.enrich(item, to);
526
+ }
527
+ }
528
+ }
529
+ }
530
+
531
+ /** Replace an entry's content, keeping its identity and tags. */
532
+ private enrich(target: MemoryItem, content: string): void {
533
+ target.content = content;
534
+ target.bookmark = content.slice(0, 80);
535
+ target.metadata.lastAccessedAt = Date.now();
536
+ }
537
+
538
+ /**
539
+ * Fold a duplicate into its survivor, refusing rewrites that drop anything.
540
+ *
541
+ * `merge` returns the model's idea of the fuller statement, and overwriting
542
+ * with it deletes whatever the model happened to leave out. The under-merge
543
+ * bias covers merging the *wrong* pair, but nothing covered a lossy rewrite
544
+ * of the right pair, so containment is checked first.
545
+ *
546
+ * @returns false when the rewrite would lose information, in which case the
547
+ * caller keeps both entries.
548
+ */
549
+ private mergeInto(
550
+ survivor: MemoryItem,
551
+ replacement: string | undefined,
552
+ duplicate: { text: string; stored: boolean },
553
+ ): boolean {
554
+ const text = replacement?.trim();
555
+ // Both statements are checked, not just the survivor: a merge has to carry
556
+ // everything either one held, so dropping the *incoming* qualifier loses it
557
+ // just as permanently as dropping the survivor's.
558
+ if (
559
+ text &&
560
+ (isLossyRewrite(survivor.content, text) || isLossyRewrite(duplicate.text, text))
561
+ ) {
562
+ return false;
563
+ }
564
+ // Remove the duplicate BEFORE rewriting the survivor: afterwards the two
565
+ // hold the same text, so removing by content would match both.
566
+ if (duplicate.stored) this.removeByContent(duplicate.text);
567
+ if (text) this.enrich(survivor, text);
568
+ return true;
569
+ }
570
+
571
+ /** Retire the entries a replacement supersedes, keeping them out of retrieval. */
572
+ private supersede(contents: string[]): void {
573
+ const targets = new Set(contents.map((c) => c.trim().toLowerCase()));
574
+ for (const map of [this.l1HotCache, this.l2WarmStore, this.l3ColdArchive]) {
575
+ for (const [id, item] of map) {
576
+ if (targets.has(item.content.trim().toLowerCase())) map.delete(id);
577
+ }
578
+ }
579
+ }
580
+
581
+ /**
582
+ * Existing memories worth adjudicating a candidate against.
583
+ *
584
+ * Cheap and deliberately over-inclusive. An identical string short-circuits
585
+ * because it is never worth a model call; anything else above the recall floor
586
+ * is offered to the adjudicator, which decides.
587
+ */
588
+ private recallSimilar(content: string): MemoryItem[] {
589
+ const needle = distinctiveTokens(content);
590
+ if (needle.size === 0) return [];
591
+ const existing = [
592
+ ...this.l1HotCache.values(),
593
+ ...this.l2WarmStore.values(),
594
+ ...this.l3ColdArchive.values(),
595
+ ];
596
+
597
+ const scored = existing
598
+ .map((item) => {
599
+ const have = distinctiveTokens(item.content);
600
+ let shared = 0;
601
+ for (const word of needle) if (have.has(word)) shared += 1;
602
+ return { item, exact: item.content.trim().toLowerCase() === content.trim().toLowerCase(), score: have.size ? shared / Math.min(needle.size, have.size) : 0 };
603
+ })
604
+ // An exact restatement is certain; the rest are merely candidates.
605
+ .filter((entry) => entry.exact || entry.score >= CANDIDATE_FLOOR)
606
+ .sort((a, b) => Number(b.exact) - Number(a.exact) || b.score - a.score);
607
+
608
+ return scored.slice(0, MAX_CANDIDATES).map((entry) => entry.item);
609
+ }
610
+
611
+ /**
612
+ * Keep the pinned tiers within budget.
613
+ *
614
+ * `maxL0Tokens`/`maxL1Tokens` were stored but never read, so L1 grew for the
615
+ * lifetime of the process. Evict least-recently-accessed first, and demote to
616
+ * L2 rather than dropping, so nothing is lost.
617
+ */
618
+ private enforceBudgets(): void {
619
+ const trim = (map: Map<string, MemoryItem>, budget: number): void => {
620
+ if (budget <= 0) return;
621
+ let total = 0;
622
+ for (const item of map.values()) total += item.content.length / 4;
623
+ if (total <= budget) return;
624
+ const ordered = [...map.values()].sort(
625
+ (a, b) => a.metadata.lastAccessedAt - b.metadata.lastAccessedAt,
626
+ );
627
+ for (const item of ordered) {
628
+ if (total <= budget) break;
629
+ map.delete(item.id);
630
+ total -= item.content.length / 4;
631
+ // Demote rather than discard: it can be promoted again later.
632
+ this.l2WarmStore.set(item.id, { ...item, tier: "L2" });
633
+ }
634
+ };
635
+ trim(this.l1HotCache, this.maxL1Tokens);
636
+ }
637
+
638
+ /**
639
+ * Add a known tension manually or from an external source.
640
+ */
641
+ addTension(tension: KnowledgeTension): void {
642
+ this.activeTensions.set(tension.id, tension);
643
+ this.stats.tensionsDetected += 1;
644
+ }
645
+
646
+ /**
647
+ * Mark a tension resolved with an optional reusable pattern.
648
+ */
649
+ resolveTension(id: string, resolution: { resolvedBy: string; pattern: string }): boolean {
650
+ const t = this.activeTensions.get(id);
651
+ if (!t) return false;
652
+ t.status = "resolved";
653
+ t.resolution = {
654
+ resolvedAt: Date.now(),
655
+ resolvedBy: resolution.resolvedBy,
656
+ pattern: resolution.pattern,
657
+ };
658
+ return true;
659
+ }
660
+
661
+ /**
662
+ * Add a memory directly to L2 warm storage or promote to L1.
663
+ */
664
+ addMemory(item: MemoryItem, targetTier: MemoryTier = "L2"): void {
665
+ item.tier = targetTier;
666
+ if (targetTier === "L1") {
667
+ this.l1HotCache.set(item.id, item);
668
+ } else if (targetTier === "L2") {
669
+ this.l2WarmStore.set(item.id, item);
670
+ } else if (targetTier === "L3") {
671
+ this.l3ColdArchive.set(item.id, item);
672
+ }
673
+ }
674
+
675
+ /**
676
+ * Drop one memory by id, from whichever tier holds it.
677
+ *
678
+ * The counterpart to `addMemory`, and the only way to retire a fact that has
679
+ * turned out to be wrong. One map is not enough to look in: a promotion or a
680
+ * demotion moves an id between them, so an id that is absent from the hot
681
+ * cache may still be in the warm store or the archive. Short-circuiting on the
682
+ * first hit is therefore safe — an id lives in exactly one of the three.
683
+ */
684
+ removeMemory(id: string): boolean {
685
+ return this.l1HotCache.delete(id) || this.l2WarmStore.delete(id) || this.l3ColdArchive.delete(id);
686
+ }
687
+
688
+ /**
689
+ * Update the proprioceptive capability record for a domain.
690
+ */
691
+ recordDomainOutcome(domain: string, success: boolean, failurePattern?: string): void {
692
+ const current = this.selfModel.domains[domain] ?? {
693
+ reliabilityScore: 0.8,
694
+ sampleCount: 0,
695
+ knownFailurePatterns: [],
696
+ recommendedStrategies: [],
697
+ };
698
+
699
+ const newSampleCount = current.sampleCount + 1;
700
+ // Moving average update
701
+ const weight = 1 / Math.min(newSampleCount, 10);
702
+ const outcomeVal = success ? 1.0 : 0.0;
703
+ const newScore = current.reliabilityScore * (1 - weight) + outcomeVal * weight;
704
+
705
+ if (!success && failurePattern && !current.knownFailurePatterns.includes(failurePattern)) {
706
+ current.knownFailurePatterns.push(failurePattern);
707
+ }
708
+
709
+ this.selfModel.domains[domain] = {
710
+ reliabilityScore: Number(newScore.toFixed(3)),
711
+ sampleCount: newSampleCount,
712
+ knownFailurePatterns: current.knownFailurePatterns,
713
+ recommendedStrategies: current.recommendedStrategies,
714
+ };
715
+ }
716
+
717
+ getSnapshot(): CognitiveMemoryStateSnapshot {
718
+ return {
719
+ l0: {
720
+ tensions: Array.from(this.activeTensions.values()),
721
+ selfModel: JSON.parse(JSON.stringify(this.selfModel)),
722
+ activeTaskTrace: this.activeTaskTrace,
723
+ },
724
+ l1: Array.from(this.l1HotCache.values()),
725
+ l2: Array.from(this.l2WarmStore.values()),
726
+ l3: Array.from(this.l3ColdArchive.values()),
727
+ stats: { ...this.stats },
728
+ };
729
+ }
730
+
731
+ loadSnapshot(snapshot: CognitiveMemoryStateSnapshot): void {
732
+ this.activeTensions.clear();
733
+ for (const t of snapshot.l0.tensions) {
734
+ this.activeTensions.set(t.id, t);
735
+ }
736
+ this.selfModel = snapshot.l0.selfModel;
737
+ this.activeTaskTrace = snapshot.l0.activeTaskTrace;
738
+
739
+ this.l1HotCache.clear();
740
+ for (const m of snapshot.l1) {
741
+ this.l1HotCache.set(m.id, m);
742
+ }
743
+
744
+ this.l2WarmStore.clear();
745
+ for (const m of snapshot.l2) {
746
+ this.l2WarmStore.set(m.id, m);
747
+ }
748
+
749
+ this.l3ColdArchive.clear();
750
+ for (const m of snapshot.l3) {
751
+ this.l3ColdArchive.set(m.id, m);
752
+ }
753
+
754
+ this.stats = { ...snapshot.stats };
755
+ }
756
+
757
+ private applyArbiterDecision(evaluation: ArbiterEvaluationResult): void {
758
+ // 1. Promotions to L1
759
+ for (const p of evaluation.promotions) {
760
+ const memory = this.l2WarmStore.get(p.memoryId) ?? this.l3ColdArchive.get(p.memoryId);
761
+ if (memory) {
762
+ memory.tier = "L1";
763
+ memory.metadata.lastAccessedAt = Date.now();
764
+ memory.metadata.accessCount += 1;
765
+ this.l1HotCache.set(memory.id, memory);
766
+ this.l2WarmStore.delete(memory.id);
767
+ this.l3ColdArchive.delete(memory.id);
768
+ }
769
+ }
770
+
771
+ // 2. Demotions to L2
772
+ for (const d of evaluation.demotions) {
773
+ const memory = this.l1HotCache.get(d.memoryId);
774
+ if (memory) {
775
+ memory.tier = "L2";
776
+ this.l2WarmStore.set(memory.id, memory);
777
+ this.l1HotCache.delete(memory.id);
778
+ }
779
+ }
780
+
781
+ // 3. New Tensions
782
+ for (const t of evaluation.detectedTensions) {
783
+ const tensionId = `tension-${Date.now()}-${Math.random().toString(36).slice(2, 6)}`;
784
+ this.addTension({
785
+ id: tensionId,
786
+ status: "active",
787
+ claimA: { source: "current conversation", statement: t.claimA, timestamp: Date.now() },
788
+ claimB: { source: "known state / files", statement: t.claimB, timestamp: Date.now() },
789
+ impact: t.impact,
790
+ taskRelevance: 1.0,
791
+ actionableQuestion: t.actionableQuestion,
792
+ });
793
+ }
794
+
795
+ // 4. Self-Model Updates
796
+ if (evaluation.selfModelUpdate) {
797
+ const { domain, success, failurePatternObserved } = evaluation.selfModelUpdate;
798
+ if (success !== undefined) {
799
+ this.recordDomainOutcome(domain, success, failurePatternObserved);
800
+ }
801
+ }
802
+ }
803
+
804
+ /**
805
+ * Fast default heuristic arbiter when an LLM arbiter model is not supplied.
806
+ */
807
+ private runHeuristicArbiter(
808
+ params: { userMessage: string; assistantResponse: string },
809
+ candidates: Array<{ id: string; bookmark: string; domains: string[] }>
810
+ ): ArbiterEvaluationResult {
811
+ const promotions: ArbiterEvaluationResult["promotions"] = [];
812
+ const demotions: ArbiterEvaluationResult["demotions"] = [];
813
+
814
+ const turn = relevanceTokens(`${params.userMessage} ${params.assistantResponse}`);
815
+ const scored = candidates
816
+ .map((candidate) => ({
817
+ candidate,
818
+ score: overlapScore(
819
+ turn,
820
+ relevanceTokens(`${candidate.bookmark} ${candidate.domains.join(" ")}`),
821
+ ),
822
+ }))
823
+ .filter((entry) => entry.score > PROMOTE_THRESHOLD)
824
+ .sort((a, b) => b.score - a.score)
825
+ .slice(0, 3);
826
+
827
+ for (const { candidate, score } of scored) {
828
+ promotions.push({
829
+ memoryId: candidate.id,
830
+ targetTier: "L1",
831
+ signalType: "anticipatory",
832
+ urgency: Math.min(1, score),
833
+ });
834
+ }
835
+
836
+ for (const item of this.l1HotCache.values()) {
837
+ const relevant =
838
+ overlapScore(turn, relevanceTokens(`${item.bookmark} ${item.metadata.domains.join(" ")}`)) > 0;
839
+ if (!relevant && this.l1HotCache.size > DEMOTE_ABOVE) {
840
+ demotions.push({
841
+ memoryId: item.id,
842
+ targetTier: "L2",
843
+ reason: "Not referenced in recent turns",
844
+ });
845
+ }
846
+ }
847
+
848
+ if (process.env.NAH_MEMORY_DEBUG) {
849
+ console.log(
850
+ ` [arbiter] scored=${JSON.stringify(scored.map((e) => ({ id: e.candidate.id, score: Number(e.score.toFixed(2)) })))} promotions=${promotions.length} demotions=${demotions.length}`,
851
+ );
852
+ }
853
+
854
+ return { promotions, demotions, pins: [], detectedTensions: [] };
855
+ }
856
+
857
+ private autoExtractTurnMemory(userMsg: string, assistantReply: string): void {
858
+ // If user provided a critical rule or config preference, store as warm memory
859
+ const preferenceMatch = userMsg.match(/(?:always|never|make sure to|remember to|use)\s+([^\.\n]+)/i);
860
+ if (preferenceMatch && preferenceMatch[1]) {
861
+ const id = `mem-pref-${Date.now()}`;
862
+ const statement = preferenceMatch[1].trim();
863
+ if (isInteractionScoped(`User preference: ${statement}`)) return;
864
+ this.addMemory({
865
+ id,
866
+ content: `User preference: ${statement}`,
867
+ bookmark: `User preference: ${statement.slice(0, 80)}`,
868
+ tier: "L1", // Pre-stage directly into hot cache
869
+ metadata: {
870
+ domains: this.selfModel.activeDomains,
871
+ createdAt: Date.now(),
872
+ lastAccessedAt: Date.now(),
873
+ accessCount: 1,
874
+ },
875
+ }, "L1");
876
+ }
877
+ }
878
+ }