stackmem 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +223 -0
- package/dist/ast-index.js +272 -0
- package/dist/cli.js +718 -0
- package/dist/config.js +2 -0
- package/dist/device.js +30 -0
- package/dist/index.js +324 -0
- package/dist/scoring.js +217 -0
- package/dist/storage.js +53 -0
- package/dist/test-scoring.js +7 -0
- package/dist/tools.js +1 -0
- package/package.json +34 -0
- package/src/ast-index.ts +349 -0
- package/src/cli.ts +814 -0
- package/src/config.ts +3 -0
- package/src/device.ts +34 -0
- package/src/index.ts +436 -0
- package/src/scoring.ts +276 -0
- package/src/storage.ts +57 -0
- package/src/test-scoring.ts +11 -0
- package/src/tools.ts +0 -0
package/src/scoring.ts
ADDED
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
// --- semantic linking ------------------------------------------------------
|
|
2
|
+
|
|
3
|
+
const STOP_WORDS = new Set([
|
|
4
|
+
"we", "use", "for", "all", "a", "the", "is", "has", "have", "are", "to",
|
|
5
|
+
"of", "in", "it", "this", "that", "with", "and", "or", "but",
|
|
6
|
+
]);
|
|
7
|
+
|
|
8
|
+
const MIN_ENTITY_LENGTH = 5;
|
|
9
|
+
|
|
10
|
+
// Key entities: words longer than 4 characters that aren't stop words.
|
|
11
|
+
// Short/common words carry little identifying signal for a memory.
|
|
12
|
+
export function extractEntities(text: string): Set<string> {
|
|
13
|
+
const entities = new Set<string>();
|
|
14
|
+
for (const word of text.toLowerCase().split(/\W+/)) {
|
|
15
|
+
if (word.length < MIN_ENTITY_LENGTH || STOP_WORDS.has(word)) continue;
|
|
16
|
+
entities.add(word);
|
|
17
|
+
}
|
|
18
|
+
return entities;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
// Containment score: shared entities over the SMALLER of the two entity
|
|
22
|
+
// sets, so a short memory whose terms are fully covered by a longer one
|
|
23
|
+
// still scores highly — unlike Jaccard, which penalizes size mismatches.
|
|
24
|
+
export function containmentScore(entitiesA: Set<string>, entitiesB: Set<string>): number {
|
|
25
|
+
const smallerSize = Math.min(entitiesA.size, entitiesB.size);
|
|
26
|
+
if (smallerSize === 0) return 0;
|
|
27
|
+
|
|
28
|
+
let shared = 0;
|
|
29
|
+
for (const entity of entitiesA) {
|
|
30
|
+
if (entitiesB.has(entity)) shared += 1;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
return shared / smallerSize;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
// --- memory links ------------------------------------------------------
|
|
37
|
+
|
|
38
|
+
export interface ExistingMemory {
|
|
39
|
+
id: string;
|
|
40
|
+
content: string;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export interface MemoryLink {
|
|
44
|
+
source_id: string;
|
|
45
|
+
target_id: string;
|
|
46
|
+
score: number;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export function computeMemoryLinks(
|
|
50
|
+
newMemoryId: string,
|
|
51
|
+
newContent: string,
|
|
52
|
+
existingMemories: ExistingMemory[],
|
|
53
|
+
threshold = 0.15
|
|
54
|
+
): MemoryLink[] {
|
|
55
|
+
const newEntities = extractEntities(newContent);
|
|
56
|
+
return existingMemories
|
|
57
|
+
.map((existing) => ({
|
|
58
|
+
source_id: newMemoryId,
|
|
59
|
+
target_id: existing.id,
|
|
60
|
+
score: containmentScore(newEntities, extractEntities(existing.content)),
|
|
61
|
+
}))
|
|
62
|
+
.filter((link) => link.score > threshold);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
// --- save-memory CRUD decision ------------------------------------------
|
|
66
|
+
|
|
67
|
+
const DUPLICATE_THRESHOLD = 0.85;
|
|
68
|
+
const CONTRADICTION_MIN = 0.4;
|
|
69
|
+
|
|
70
|
+
// Words that signal the new memory is meant to replace an old one rather
|
|
71
|
+
// than just relate to it (e.g. "we switched from X to Y").
|
|
72
|
+
const NEGATION_WORDS = [
|
|
73
|
+
"switched", "removed", "replaced", "no longer", "instead",
|
|
74
|
+
"deprecated", "reverted", "changed",
|
|
75
|
+
];
|
|
76
|
+
|
|
77
|
+
function containsNegationWord(text: string): boolean {
|
|
78
|
+
const lower = text.toLowerCase();
|
|
79
|
+
return NEGATION_WORDS.some((word) => lower.includes(word));
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
export type SaveAction =
|
|
83
|
+
| { action: "duplicate"; matchedId: string }
|
|
84
|
+
| { action: "superseded"; oldId: string }
|
|
85
|
+
| { action: "created" };
|
|
86
|
+
|
|
87
|
+
// Decides whether a new memory is a near-duplicate of an existing one
|
|
88
|
+
// (skip it), supersedes an existing one (resolve the old, insert the new),
|
|
89
|
+
// or is genuinely new. A negation word (e.g. "switched", "no longer")
|
|
90
|
+
// reroutes an otherwise-duplicate-level score into a supersede instead —
|
|
91
|
+
// without that carve-out, a "switched from X to Y" memory that repeats X's
|
|
92
|
+
// keywords would score as a duplicate of the memory it's meant to replace.
|
|
93
|
+
// Scanning finds the best-scoring match in each band rather than stopping
|
|
94
|
+
// at the first hit, so the result doesn't depend on row order.
|
|
95
|
+
export function decideSaveAction(
|
|
96
|
+
newContent: string,
|
|
97
|
+
existingMemories: ExistingMemory[]
|
|
98
|
+
): SaveAction {
|
|
99
|
+
const newEntities = extractEntities(newContent);
|
|
100
|
+
const hasNegation = containsNegationWord(newContent);
|
|
101
|
+
|
|
102
|
+
let bestDuplicate: { id: string; score: number } | null = null;
|
|
103
|
+
let bestSupersede: { id: string; score: number } | null = null;
|
|
104
|
+
|
|
105
|
+
for (const existing of existingMemories) {
|
|
106
|
+
const score = containmentScore(newEntities, extractEntities(existing.content));
|
|
107
|
+
|
|
108
|
+
if (score > DUPLICATE_THRESHOLD) {
|
|
109
|
+
if (hasNegation) {
|
|
110
|
+
if (!bestSupersede || score > bestSupersede.score) {
|
|
111
|
+
bestSupersede = { id: existing.id, score };
|
|
112
|
+
}
|
|
113
|
+
} else if (!bestDuplicate || score > bestDuplicate.score) {
|
|
114
|
+
bestDuplicate = { id: existing.id, score };
|
|
115
|
+
}
|
|
116
|
+
} else if (hasNegation && score >= CONTRADICTION_MIN) {
|
|
117
|
+
if (!bestSupersede || score > bestSupersede.score) {
|
|
118
|
+
bestSupersede = { id: existing.id, score };
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
if (bestDuplicate) return { action: "duplicate", matchedId: bestDuplicate.id };
|
|
124
|
+
if (bestSupersede) return { action: "superseded", oldId: bestSupersede.id };
|
|
125
|
+
return { action: "created" };
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// --- memory decay --------------------------------------------------------
|
|
129
|
+
|
|
130
|
+
// Decay constant tuned for a ~14-day half-life: exp(-0.05 * 14) ≈ 0.5.
|
|
131
|
+
const DECAY_LAMBDA = 0.05;
|
|
132
|
+
|
|
133
|
+
// Recency-weighted relevance: frequently-accessed memories decay slower,
|
|
134
|
+
// but every memory still fades over time regardless of access count.
|
|
135
|
+
export function computeDecayScore(createdAt: Date, accessCount: number): number {
|
|
136
|
+
const daysSinceCreated = (Date.now() - createdAt.getTime()) / (1000 * 60 * 60 * 24);
|
|
137
|
+
return (1 + Math.log(1 + accessCount)) * Math.exp(-DECAY_LAMBDA * daysSinceCreated);
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
// --- context budget ------------------------------------------------------
|
|
141
|
+
|
|
142
|
+
// Rough token estimate: ~4 characters per token.
|
|
143
|
+
export function estimateTokens(text: string): number {
|
|
144
|
+
return Math.ceil(text.length / 4);
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
// Takes a decay-sorted memory array (highest relevance first) and keeps
|
|
148
|
+
// adding memories until the next one would push the running token count
|
|
149
|
+
// over budget, so what gets injected into an agent's context stays capped.
|
|
150
|
+
export function applyContextBudget(memories: any[], budgetTokens: number): any[] {
|
|
151
|
+
const result: any[] = [];
|
|
152
|
+
let usedTokens = 0;
|
|
153
|
+
|
|
154
|
+
for (const memory of memories) {
|
|
155
|
+
const tokens = estimateTokens(memory.content);
|
|
156
|
+
if (usedTokens + tokens > budgetTokens) break;
|
|
157
|
+
result.push(memory);
|
|
158
|
+
usedTokens += tokens;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
return result;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
// --- memory compression --------------------------------------------------
|
|
165
|
+
|
|
166
|
+
const CLUSTER_THRESHOLD = 0.2;
|
|
167
|
+
|
|
168
|
+
// Single-linkage clustering over containment score: two memories join the
|
|
169
|
+
// same cluster if they score > 0.2, and clusters merge transitively (via
|
|
170
|
+
// union-find) even if the two memories that join them don't directly score
|
|
171
|
+
// above the threshold themselves.
|
|
172
|
+
export function clusterMemories(memories: any[]): any[][] {
|
|
173
|
+
const parent = memories.map((_, index) => index);
|
|
174
|
+
|
|
175
|
+
function find(index: number): number {
|
|
176
|
+
while (parent[index] !== index) {
|
|
177
|
+
parent[index] = parent[parent[index]];
|
|
178
|
+
index = parent[index];
|
|
179
|
+
}
|
|
180
|
+
return index;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
function union(a: number, b: number) {
|
|
184
|
+
const rootA = find(a);
|
|
185
|
+
const rootB = find(b);
|
|
186
|
+
if (rootA !== rootB) parent[rootA] = rootB;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
const entities = memories.map((memory) => extractEntities(memory.content));
|
|
190
|
+
|
|
191
|
+
for (let i = 0; i < memories.length; i++) {
|
|
192
|
+
for (let j = i + 1; j < memories.length; j++) {
|
|
193
|
+
if (containmentScore(entities[i], entities[j]) > CLUSTER_THRESHOLD) {
|
|
194
|
+
union(i, j);
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
const clustersByRoot = new Map<number, any[]>();
|
|
200
|
+
for (let i = 0; i < memories.length; i++) {
|
|
201
|
+
const root = find(i);
|
|
202
|
+
if (!clustersByRoot.has(root)) clustersByRoot.set(root, []);
|
|
203
|
+
clustersByRoot.get(root)!.push(memories[i]);
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
return Array.from(clustersByRoot.values());
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
const SUMMARY_ITEM_LENGTH = 60;
|
|
210
|
+
|
|
211
|
+
// Builds a compact summary for a cluster of related memories: the most
|
|
212
|
+
// common type in the cluster, the most frequent shared entity as the
|
|
213
|
+
// "topic", and each memory's content truncated and joined.
|
|
214
|
+
export function buildSummaryContent(cluster: any[]): string {
|
|
215
|
+
const typeCounts = new Map<string, number>();
|
|
216
|
+
for (const memory of cluster) {
|
|
217
|
+
typeCounts.set(memory.type, (typeCounts.get(memory.type) ?? 0) + 1);
|
|
218
|
+
}
|
|
219
|
+
let mostCommonType = cluster[0]?.type ?? "discovery";
|
|
220
|
+
let maxTypeCount = 0;
|
|
221
|
+
for (const [type, count] of typeCounts) {
|
|
222
|
+
if (count > maxTypeCount) {
|
|
223
|
+
maxTypeCount = count;
|
|
224
|
+
mostCommonType = type;
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
const entityCounts = new Map<string, number>();
|
|
229
|
+
for (const memory of cluster) {
|
|
230
|
+
for (const entity of extractEntities(memory.content)) {
|
|
231
|
+
entityCounts.set(entity, (entityCounts.get(entity) ?? 0) + 1);
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
let topic = "general";
|
|
235
|
+
let maxEntityCount = 0;
|
|
236
|
+
for (const [entity, count] of entityCounts) {
|
|
237
|
+
if (count > maxEntityCount) {
|
|
238
|
+
maxEntityCount = count;
|
|
239
|
+
topic = entity;
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
const items = cluster
|
|
244
|
+
.map((memory) => memory.content.slice(0, SUMMARY_ITEM_LENGTH))
|
|
245
|
+
.join("; ");
|
|
246
|
+
|
|
247
|
+
return `[${mostCommonType.toUpperCase()} cluster] ${topic}: ${items}`;
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
// --- task relevance --------------------------------------------------------
|
|
251
|
+
|
|
252
|
+
function splitWords(text: string): string[] {
|
|
253
|
+
return text
|
|
254
|
+
.toLowerCase()
|
|
255
|
+
.split(/\s+/)
|
|
256
|
+
.filter((word) => word.length > 0 && !STOP_WORDS.has(word));
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
// Blends entity containment (structural overlap) with raw word overlap
|
|
260
|
+
// (surface-level phrasing match) so a memory that shares the task's exact
|
|
261
|
+
// wording scores well even if its extracted entities don't line up neatly.
|
|
262
|
+
export function scoreMemoryRelevance(memoryContent: string, task: string): number {
|
|
263
|
+
const containment = containmentScore(extractEntities(memoryContent), extractEntities(task));
|
|
264
|
+
|
|
265
|
+
const memoryWords = new Set(splitWords(memoryContent));
|
|
266
|
+
const taskWords = splitWords(task);
|
|
267
|
+
|
|
268
|
+
let shared = 0;
|
|
269
|
+
for (const word of taskWords) {
|
|
270
|
+
if (memoryWords.has(word)) shared++;
|
|
271
|
+
}
|
|
272
|
+
const wordOverlapRatio = taskWords.length === 0 ? 0 : shared / taskWords.length;
|
|
273
|
+
|
|
274
|
+
return containment * 0.7 + wordOverlapRatio * 0.3;
|
|
275
|
+
}
|
|
276
|
+
// TODO: optimize the clustering algorithm
|
package/src/storage.ts
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
// Run these in the Supabase SQL editor after changing the RLS scoping
|
|
2
|
+
// mechanism from a session-config RPC to a request header. `set_config` from
|
|
3
|
+
// an RPC call doesn't persist to the next request under PostgREST (each call
|
|
4
|
+
// is its own transaction), so RLS now reads the 'x-device-id' header that
|
|
5
|
+
// supabase-js attaches to every request instead.
|
|
6
|
+
//
|
|
7
|
+
// DROP POLICY IF EXISTS "device owns memories" ON memories;
|
|
8
|
+
// CREATE POLICY "device owns memories" ON memories
|
|
9
|
+
// USING (device_id = (current_setting('request.headers', true)::json->>'x-device-id'))
|
|
10
|
+
// WITH CHECK (device_id = (current_setting('request.headers', true)::json->>'x-device-id'));
|
|
11
|
+
//
|
|
12
|
+
// DROP POLICY IF EXISTS "device owns execution_log" ON execution_log;
|
|
13
|
+
// CREATE POLICY "device owns execution_log" ON execution_log
|
|
14
|
+
// USING (device_id = (current_setting('request.headers', true)::json->>'x-device-id'))
|
|
15
|
+
// WITH CHECK (device_id = (current_setting('request.headers', true)::json->>'x-device-id'));
|
|
16
|
+
//
|
|
17
|
+
// DROP POLICY IF EXISTS "device owns memory_links" ON memory_links;
|
|
18
|
+
// CREATE POLICY "device owns memory_links" ON memory_links
|
|
19
|
+
// USING (
|
|
20
|
+
// source_id in (
|
|
21
|
+
// select id from memories
|
|
22
|
+
// where device_id = (current_setting('request.headers', true)::json->>'x-device-id')
|
|
23
|
+
// )
|
|
24
|
+
// );
|
|
25
|
+
|
|
26
|
+
import { createClient, type SupabaseClient } from "@supabase/supabase-js";
|
|
27
|
+
import { SUPABASE_ANON_KEY, SUPABASE_URL } from "./config.js";
|
|
28
|
+
import { getDeviceId } from "./device.js";
|
|
29
|
+
|
|
30
|
+
export const supabase: SupabaseClient = createClient(SUPABASE_URL, SUPABASE_ANON_KEY, {
|
|
31
|
+
global: {
|
|
32
|
+
headers: {
|
|
33
|
+
"x-device-id": getDeviceId(),
|
|
34
|
+
},
|
|
35
|
+
},
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
// Run this migration in the Supabase SQL editor to add the access_count
|
|
39
|
+
// column used by the memory decay scoring in src/scoring.ts (computeDecayScore)
|
|
40
|
+
// and consumed by cm start / session_start in src/cli.ts and src/index.ts.
|
|
41
|
+
//
|
|
42
|
+
// ALTER TABLE memories ADD COLUMN IF NOT EXISTS access_count integer default 0;
|
|
43
|
+
|
|
44
|
+
// Run this migration in the Supabase SQL editor to create the memory_links
|
|
45
|
+
// table used by the semantic linking system in src/index.ts (save_memory /
|
|
46
|
+
// session_start). Assumes memories.id is a uuid, matching Supabase's default.
|
|
47
|
+
//
|
|
48
|
+
// create table memory_links (
|
|
49
|
+
// id uuid primary key default gen_random_uuid(),
|
|
50
|
+
// source_id uuid not null references memories(id) on delete cascade,
|
|
51
|
+
// target_id uuid not null references memories(id) on delete cascade,
|
|
52
|
+
// score double precision not null,
|
|
53
|
+
// created_at timestamptz not null default now()
|
|
54
|
+
// );
|
|
55
|
+
//
|
|
56
|
+
// create index memory_links_source_id_idx on memory_links(source_id);
|
|
57
|
+
// create index memory_links_target_id_idx on memory_links(target_id);
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import { extractEntities, containmentScore } from "./scoring.js";
|
|
2
|
+
|
|
3
|
+
const textA = "we use Supabase for database storage";
|
|
4
|
+
const textB = "Supabase free tier has 500MB limit";
|
|
5
|
+
|
|
6
|
+
const entitiesA = extractEntities(textA);
|
|
7
|
+
const entitiesB = extractEntities(textB);
|
|
8
|
+
|
|
9
|
+
const score = containmentScore(entitiesA, entitiesB);
|
|
10
|
+
|
|
11
|
+
console.log(`SCORE: [${textA}] vs [${textB}] = ${score}`);
|
package/src/tools.ts
ADDED
|
File without changes
|