claude-memory-admin 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/stats.mjs ADDED
@@ -0,0 +1,203 @@
1
+ // Size accounting and overlap detection for a project's memory.
2
+ //
3
+ // The point of all of this is MEMORY.md: it is read into context at the start of
4
+ // every session, so its size is a recurring cost in a way the individual memory
5
+ // files (read only when followed) are not. Everything here exists to make that
6
+ // cost visible and to point at what to prune.
7
+
8
+ /**
9
+ * Rough token estimate. Deliberately a heuristic - roughly four characters per
10
+ * token for English prose - because the real tokeniser is not available here.
11
+ * Shown as context only; the limits that actually bite are lines and bytes.
12
+ */
13
+ export function estimateTokens(text) {
14
+ if (!text) return 0;
15
+ return Math.round(text.length / 4);
16
+ }
17
+
18
+ // The limits Claude Code actually enforces on MEMORY.md: the first 200 lines or
19
+ // 25KB, whichever comes first, are loaded at the start of a session. Anything
20
+ // past that is silently dropped on the next load, which is the failure mode this
21
+ // meter exists to prevent.
22
+ export const INDEX_LINE_LIMIT = 200;
23
+ export const INDEX_BYTE_LIMIT = 25 * 1024;
24
+
25
+ /** Hooks longer than this dominate the index; they are the usual bloat. */
26
+ export const LONG_HOOK_CHARS = 200;
27
+
28
+ /**
29
+ * The text that actually counts against the limits. Claude Code strips YAML
30
+ * frontmatter and block-level HTML comments before loading the index, so
31
+ * measuring the raw file would overstate the size.
32
+ */
33
+ export function loadedIndexText(indexText) {
34
+ if (!indexText) return '';
35
+ let text = indexText;
36
+ const frontmatter = text.match(/^---\n[\s\S]*?\n---\n?/);
37
+ if (frontmatter) text = text.slice(frontmatter[0].length);
38
+ return text.replace(/^[ \t]*<!--[\s\S]*?-->[ \t]*\n?/gm, '');
39
+ }
40
+
41
+ export function indexStats(indexText, entries) {
42
+ if (indexText === null || indexText === undefined) {
43
+ return {
44
+ bytes: 0, lines: 0, tokens: 0, entryCount: 0, longHooks: [], longestHook: 0,
45
+ lineLimit: INDEX_LINE_LIMIT, byteLimit: INDEX_BYTE_LIMIT,
46
+ linePercent: 0, bytePercent: 0, worstPercent: 0, level: 'ok', overLimit: false, nearLimit: false,
47
+ };
48
+ }
49
+
50
+ const loaded = loadedIndexText(indexText);
51
+ const lines = loaded.split('\n').filter((line, i, all) => i < all.length - 1 || line.length > 0).length;
52
+ const bytes = Buffer.byteLength(loaded, 'utf8');
53
+
54
+ const longHooks = entries
55
+ .filter((entry) => entry.hook.length > LONG_HOOK_CHARS)
56
+ .map((entry) => ({ index: entry.index, file: entry.file, title: entry.title, hookLength: entry.hook.length }))
57
+ .sort((a, b) => b.hookLength - a.hookLength);
58
+
59
+ const linePercent = (lines / INDEX_LINE_LIMIT) * 100;
60
+ const bytePercent = (bytes / INDEX_BYTE_LIMIT) * 100;
61
+ const worstPercent = Math.max(linePercent, bytePercent);
62
+
63
+ return {
64
+ bytes,
65
+ lines,
66
+ rawBytes: Buffer.byteLength(indexText, 'utf8'),
67
+ tokens: estimateTokens(loaded),
68
+ entryCount: entries.length,
69
+ longHooks,
70
+ longestHook: entries.reduce((max, e) => Math.max(max, e.hook.length), 0),
71
+ lineLimit: INDEX_LINE_LIMIT,
72
+ byteLimit: INDEX_BYTE_LIMIT,
73
+ linePercent,
74
+ bytePercent,
75
+ worstPercent,
76
+ limitedBy: linePercent >= bytePercent ? 'lines' : 'bytes',
77
+ overLimit: lines > INDEX_LINE_LIMIT || bytes > INDEX_BYTE_LIMIT,
78
+ nearLimit: worstPercent >= 75,
79
+ level: worstPercent > 100 ? 'over' : worstPercent >= 75 ? 'near' : 'ok',
80
+ };
81
+ }
82
+
83
+ function words(text) {
84
+ return String(text || '')
85
+ .toLowerCase()
86
+ .split(/[^a-z0-9]+/)
87
+ // Two-character tokens are kept on purpose: "v4", "nx", "ci" and friends
88
+ // are among the most discriminating words in these names.
89
+ .filter((word) => word.length > 1 && !STOP_WORDS.has(word));
90
+ }
91
+
92
+ const STOP_WORDS = new Set([
93
+ 'the', 'and', 'for', 'not', 'but', 'with', 'this', 'that', 'from', 'into', 'via', 'are', 'was',
94
+ 'has', 'have', 'its', 'you', 'use', 'used', 'uses', 'must', 'need', 'needs', 'when', 'only',
95
+ 'per', 'all', 'any', 'own', 'one', 'two', 'new', 'old', 'now', 'why', 'how', 'never', 'always',
96
+ ]);
97
+
98
+ /**
99
+ * Cosine similarity over word vectors weighted by how rare each word is inside
100
+ * this project.
101
+ *
102
+ * Plain character trigrams were tried first and were wrong here: memory names
103
+ * share long prefixes by convention (argus-mobile-*, admincenter-*), so unrelated
104
+ * notes scored as high as genuinely overlapping ones. Weighting by rarity fixes
105
+ * that - a word every second memory contains says nothing about overlap, while
106
+ * a word shared by only two is a strong signal.
107
+ */
108
+ function buildIdf(documents) {
109
+ const frequency = new Map();
110
+ for (const document of documents) {
111
+ for (const word of new Set(document)) {
112
+ frequency.set(word, (frequency.get(word) || 0) + 1);
113
+ }
114
+ }
115
+ const total = Math.max(documents.length, 1);
116
+ const idf = new Map();
117
+ for (const [word, count] of frequency) {
118
+ idf.set(word, Math.log((total + 1) / (count + 0.5)));
119
+ }
120
+ return idf;
121
+ }
122
+
123
+ function vector(tokens, idf) {
124
+ const counts = new Map();
125
+ for (const word of tokens) counts.set(word, (counts.get(word) || 0) + 1);
126
+ const out = new Map();
127
+ let norm = 0;
128
+ for (const [word, count] of counts) {
129
+ const weight = (idf.get(word) ?? 1) * (1 + Math.log(count));
130
+ out.set(word, weight);
131
+ norm += weight * weight;
132
+ }
133
+ return { out, norm: Math.sqrt(norm) || 1 };
134
+ }
135
+
136
+ function cosine(left, right) {
137
+ let dot = 0;
138
+ const [small, large] = left.out.size <= right.out.size ? [left, right] : [right, left];
139
+ for (const [word, weight] of small.out) {
140
+ const other = large.out.get(word);
141
+ if (other) dot += weight * other;
142
+ }
143
+ return dot / (left.norm * right.norm);
144
+ }
145
+
146
+ /** Similarity of two short texts against a corpus, 0 to 1. */
147
+ export function similarity(a, b, corpus = [a, b]) {
148
+ const idf = buildIdf(corpus.map(words));
149
+ return cosine(vector(words(a), idf), vector(words(b), idf));
150
+ }
151
+
152
+ // There is no score that cleanly separates "duplicate" from "unrelated" here,
153
+ // and pretending otherwise would just produce confident nonsense. The UI shows a
154
+ // ranked shortlist above a modest floor and calls it possible overlap.
155
+ export const DUPLICATE_THRESHOLD = 0.18;
156
+ export const DUPLICATE_LIMIT = 12;
157
+
158
+ /**
159
+ * Pairs of memories whose name and description overlap enough to be worth a
160
+ * look. Compares the identifying text only, not the bodies: two notes about the
161
+ * same subject are the thing to catch, and full bodies drown that signal in
162
+ * shared vocabulary.
163
+ */
164
+ export function findDuplicates(memories, threshold = DUPLICATE_THRESHOLD) {
165
+ const keyed = memories.map((memory) => ({
166
+ file: memory.file,
167
+ name: memory.name,
168
+ description: memory.description,
169
+ tokens: words(`${memory.name} ${memory.description}`),
170
+ }));
171
+ if (keyed.length < 2) return [];
172
+
173
+ const idf = buildIdf(keyed.map((k) => k.tokens));
174
+ const vectors = keyed.map((k) => vector(k.tokens, idf));
175
+
176
+ const pairs = [];
177
+ for (let i = 0; i < keyed.length; i++) {
178
+ for (let j = i + 1; j < keyed.length; j++) {
179
+ const score = cosine(vectors[i], vectors[j]);
180
+ if (score >= threshold) {
181
+ const shared = [...vectors[i].out.keys()]
182
+ .filter((word) => vectors[j].out.has(word))
183
+ .sort((a, b) => (idf.get(b) ?? 0) - (idf.get(a) ?? 0))
184
+ .slice(0, 5);
185
+ pairs.push({
186
+ a: { file: keyed[i].file, name: keyed[i].name, description: keyed[i].description },
187
+ b: { file: keyed[j].file, name: keyed[j].name, description: keyed[j].description },
188
+ score: Math.round(score * 100),
189
+ shared,
190
+ });
191
+ }
192
+ }
193
+ }
194
+ return pairs.sort((x, y) => y.score - x.score).slice(0, DUPLICATE_LIMIT);
195
+ }
196
+
197
+ /** Age in whole days, or null when nothing recorded a date. */
198
+ export function ageInDays(iso, now = Date.now()) {
199
+ if (!iso) return null;
200
+ const then = Date.parse(iso);
201
+ if (Number.isNaN(then)) return null;
202
+ return Math.max(0, Math.floor((now - then) / 86400000));
203
+ }