minnimemory 1.0.0-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +39 -0
- package/README.md +824 -0
- package/dist/bench.d.ts +98 -0
- package/dist/bench.js +142 -0
- package/dist/benchReport.d.ts +12 -0
- package/dist/benchReport.js +128 -0
- package/dist/bounds.d.ts +40 -0
- package/dist/bounds.js +44 -0
- package/dist/cli.d.ts +15 -0
- package/dist/cli.js +503 -0
- package/dist/compile.d.ts +187 -0
- package/dist/compile.js +516 -0
- package/dist/discover.d.ts +125 -0
- package/dist/discover.js +520 -0
- package/dist/doctor.d.ts +9 -0
- package/dist/doctor.js +67 -0
- package/dist/episodic.d.ts +47 -0
- package/dist/episodic.js +130 -0
- package/dist/hook.d.ts +45 -0
- package/dist/hook.js +104 -0
- package/dist/index.d.ts +18 -0
- package/dist/index.js +18 -0
- package/dist/init.d.ts +125 -0
- package/dist/init.js +475 -0
- package/dist/instructions.d.ts +60 -0
- package/dist/instructions.js +270 -0
- package/dist/mcp.d.ts +109 -0
- package/dist/mcp.js +252 -0
- package/dist/mcpServer.d.ts +136 -0
- package/dist/mcpServer.js +997 -0
- package/dist/paths.d.ts +25 -0
- package/dist/paths.js +47 -0
- package/dist/recall.d.ts +113 -0
- package/dist/recall.js +256 -0
- package/dist/recallDir.d.ts +50 -0
- package/dist/recallDir.js +187 -0
- package/dist/reorganize.d.ts +62 -0
- package/dist/reorganize.js +216 -0
- package/dist/report.d.ts +16 -0
- package/dist/report.js +204 -0
- package/dist/router.d.ts +141 -0
- package/dist/router.js +314 -0
- package/dist/rules.d.ts +32 -0
- package/dist/rules.js +651 -0
- package/dist/scan.d.ts +110 -0
- package/dist/scan.js +173 -0
- package/dist/text.d.ts +158 -0
- package/dist/text.js +395 -0
- package/dist/tokenizer.d.ts +26 -0
- package/dist/tokenizer.js +69 -0
- package/dist/types.d.ts +156 -0
- package/dist/types.js +17 -0
- package/dist/version.d.ts +7 -0
- package/dist/version.js +7 -0
- package/dist/writeProtocol.d.ts +19 -0
- package/dist/writeProtocol.js +45 -0
- package/examples/CLAUDE.md +75 -0
- package/examples/README.md +7 -0
- package/package.json +52 -0
package/dist/rules.js
ADDED
|
@@ -0,0 +1,651 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The MM001-MM010 rule set.
|
|
3
|
+
*
|
|
4
|
+
* Every rule is pure: it reads a Workspace and returns Findings. Nothing here touches the
|
|
5
|
+
* filesystem or the network, which is what makes the whole engine testable from fixtures.
|
|
6
|
+
*/
|
|
7
|
+
import fs from "node:fs";
|
|
8
|
+
import path from "node:path";
|
|
9
|
+
import { COMPILED_DIR_NAME, driftHash, verbatimHash } from "./compile.js";
|
|
10
|
+
import { blocks, GRANULARITY_TOKENS, isEpisodicBody, normalise, sections, sectionVolatilityReasons, stripGeneratedRegions, } from "./text.js";
|
|
11
|
+
import { estimateTokens, formatTokens } from "./tokenizer.js";
|
|
12
|
+
import { alwaysLoaded } from "./types.js";
|
|
13
|
+
// ---------------------------------------------------------------------------
|
|
14
|
+
// shared helpers
|
|
15
|
+
// ---------------------------------------------------------------------------
|
|
16
|
+
/** Sections of a memory file, adapted to the MemoryFile wrapper. */
|
|
17
|
+
function fileSections(file) {
|
|
18
|
+
return sections(file.lines);
|
|
19
|
+
}
|
|
20
|
+
function sectionText(file, s) {
|
|
21
|
+
return file.lines.slice(s.startLine - 1, s.endLine).join("\n");
|
|
22
|
+
}
|
|
23
|
+
function fileBlocks(file) {
|
|
24
|
+
return blocks(file.lines);
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* Rules return every finding they have. Truncation is a display concern and lives in the human
|
|
28
|
+
* renderer (report.ts), so `--json` and the exit code always see the complete list: capping here
|
|
29
|
+
* made the overflow marker's own "re-run with --json for the full list" hint false, and put the
|
|
30
|
+
* findings it hid out of reach of every output mode.
|
|
31
|
+
*/
|
|
32
|
+
// ---------------------------------------------------------------------------
|
|
33
|
+
// MM001 always-loaded memory exceeds the token budget
|
|
34
|
+
// ---------------------------------------------------------------------------
|
|
35
|
+
const mm001 = {
|
|
36
|
+
id: "MM001",
|
|
37
|
+
severity: "high",
|
|
38
|
+
title: "always-loaded memory exceeds the token budget",
|
|
39
|
+
run({ workspace, config }) {
|
|
40
|
+
const loaded = alwaysLoaded(workspace);
|
|
41
|
+
const total = loaded.reduce((n, f) => n + f.tokens, 0);
|
|
42
|
+
if (total <= config.budget)
|
|
43
|
+
return [];
|
|
44
|
+
const biggest = [...loaded].sort((a, b) => b.tokens - a.tokens)[0];
|
|
45
|
+
const where = loaded.length === 1
|
|
46
|
+
? `${biggest?.rel} is ${formatTokens(total)} tokens`
|
|
47
|
+
: `${loaded.length} always-loaded files total ${formatTokens(total)} tokens`;
|
|
48
|
+
return [
|
|
49
|
+
{
|
|
50
|
+
rule: "MM001",
|
|
51
|
+
severity: "high",
|
|
52
|
+
message: `${where}, resent every turn`,
|
|
53
|
+
file: biggest?.rel ?? ".",
|
|
54
|
+
tokens: total,
|
|
55
|
+
hint: `budget is ${formatTokens(config.budget)}; move task-specific content into routed OnDemandMemory files`,
|
|
56
|
+
},
|
|
57
|
+
];
|
|
58
|
+
},
|
|
59
|
+
};
|
|
60
|
+
// ---------------------------------------------------------------------------
|
|
61
|
+
// MM002 content derivable from the repo itself
|
|
62
|
+
// ---------------------------------------------------------------------------
|
|
63
|
+
const TREE_CHAR = /[│├└┌┐┘┬┴┼]/;
|
|
64
|
+
/**
|
|
65
|
+
* One line of a directory listing, in any of the shapes real memory files use:
|
|
66
|
+
* a bare path, a tree-drawn path, either of those with a trailing "# comment" or "- comment",
|
|
67
|
+
* or a table row whose first cell is a path. Found by the 2026-09-02 sweep: the most common
|
|
68
|
+
* real-world shape is a Markdown table of directories, which the old bare-path pattern missed.
|
|
69
|
+
*/
|
|
70
|
+
const PATH_TOKEN = /`?([\w.@+\-<>]+(?:[/\\][\w.@+\-<>*]*)*[/\\]?)`?/;
|
|
71
|
+
// No `^\s*` before a class that also contains `\s`: two adjacent whitespace quantifiers made the
|
|
72
|
+
// old PATHY pattern quadratic on a line of spaces (security audit 2026-09-02, finding 3).
|
|
73
|
+
const TREE_LINE = new RegExp(`^[│├└─|\\s]*${PATH_TOKEN.source}\\s*(?:(?:#|//|-|—|:)\\s.*)?$`);
|
|
74
|
+
const TABLE_ROW = new RegExp(`^\\s*\\|\\s*${PATH_TOKEN.source}\\s*\\|`);
|
|
75
|
+
/** Does the path token look like a path rather than a plain word: a slash, or a file extension. */
|
|
76
|
+
function pathLike(token) {
|
|
77
|
+
return /[/\\]/.test(token) || /\.[a-z]{1,5}$/i.test(token);
|
|
78
|
+
}
|
|
79
|
+
function treeRuns(file) {
|
|
80
|
+
const out = [];
|
|
81
|
+
let start = -1;
|
|
82
|
+
let treeChars = 0;
|
|
83
|
+
let paths = 0;
|
|
84
|
+
let tableRows = 0;
|
|
85
|
+
const close = (end) => {
|
|
86
|
+
if (start >= 0) {
|
|
87
|
+
const len = end - start + 1;
|
|
88
|
+
if (len >= 6 && (treeChars > 0 || paths >= 3)) {
|
|
89
|
+
const text = file.lines.slice(start - 1, end).join("\n");
|
|
90
|
+
const what = tableRows > len / 2 ? "a directory listing" : "a directory tree";
|
|
91
|
+
out.push({
|
|
92
|
+
rule: "MM002",
|
|
93
|
+
severity: "med",
|
|
94
|
+
message: `lines ${start}-${end} are ${what} derivable from the repo itself`,
|
|
95
|
+
file: file.rel,
|
|
96
|
+
startLine: start,
|
|
97
|
+
endLine: end,
|
|
98
|
+
tokens: estimateTokens(text),
|
|
99
|
+
hint: "delete it; the agent can list the directory, and this copy goes stale silently",
|
|
100
|
+
});
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
start = -1;
|
|
104
|
+
treeChars = 0;
|
|
105
|
+
paths = 0;
|
|
106
|
+
tableRows = 0;
|
|
107
|
+
};
|
|
108
|
+
file.lines.forEach((line, i) => {
|
|
109
|
+
const n = i + 1;
|
|
110
|
+
if (line.trim() === "")
|
|
111
|
+
return close(i);
|
|
112
|
+
const row = TABLE_ROW.exec(line);
|
|
113
|
+
const tree = row ? null : TREE_LINE.exec(line);
|
|
114
|
+
const token = row?.[1] ?? tree?.[1];
|
|
115
|
+
const isTree = tree !== null && tree !== undefined && TREE_CHAR.test(line);
|
|
116
|
+
if (token && (isTree || pathLike(token))) {
|
|
117
|
+
if (start < 0)
|
|
118
|
+
start = n;
|
|
119
|
+
if (isTree)
|
|
120
|
+
treeChars++;
|
|
121
|
+
if (pathLike(token))
|
|
122
|
+
paths++;
|
|
123
|
+
if (row)
|
|
124
|
+
tableRows++;
|
|
125
|
+
}
|
|
126
|
+
else {
|
|
127
|
+
close(i);
|
|
128
|
+
}
|
|
129
|
+
});
|
|
130
|
+
close(file.lines.length);
|
|
131
|
+
return out;
|
|
132
|
+
}
|
|
133
|
+
function duplicatedScripts(file, root) {
|
|
134
|
+
const pkgPath = path.join(root, "package.json");
|
|
135
|
+
if (!fs.existsSync(pkgPath))
|
|
136
|
+
return [];
|
|
137
|
+
let names = [];
|
|
138
|
+
try {
|
|
139
|
+
const pkg = JSON.parse(fs.readFileSync(pkgPath, "utf8"));
|
|
140
|
+
names = Object.keys(pkg.scripts ?? {});
|
|
141
|
+
}
|
|
142
|
+
catch {
|
|
143
|
+
return [];
|
|
144
|
+
}
|
|
145
|
+
if (names.length === 0)
|
|
146
|
+
return [];
|
|
147
|
+
const hits = names.filter((n) => file.content.includes(`npm run ${n}`));
|
|
148
|
+
if (hits.length < 3)
|
|
149
|
+
return [];
|
|
150
|
+
return [
|
|
151
|
+
{
|
|
152
|
+
rule: "MM002",
|
|
153
|
+
severity: "med",
|
|
154
|
+
message: `${hits.length} npm scripts are restated here and already exist in package.json`,
|
|
155
|
+
file: file.rel,
|
|
156
|
+
hint: "point the agent at package.json instead of copying the script list",
|
|
157
|
+
},
|
|
158
|
+
];
|
|
159
|
+
}
|
|
160
|
+
const mm002 = {
|
|
161
|
+
id: "MM002",
|
|
162
|
+
severity: "med",
|
|
163
|
+
title: "content derivable from the repo itself",
|
|
164
|
+
run({ workspace }) {
|
|
165
|
+
const out = [];
|
|
166
|
+
for (const file of workspace.files) {
|
|
167
|
+
out.push(...treeRuns(file));
|
|
168
|
+
if (file.alwaysLoaded)
|
|
169
|
+
out.push(...duplicatedScripts(file, workspace.root));
|
|
170
|
+
}
|
|
171
|
+
return out.sort((a, b) => (b.tokens ?? 0) - (a.tokens ?? 0));
|
|
172
|
+
},
|
|
173
|
+
};
|
|
174
|
+
// ---------------------------------------------------------------------------
|
|
175
|
+
// MM003 volatile content in the always-loaded prefix
|
|
176
|
+
// ---------------------------------------------------------------------------
|
|
177
|
+
const mm003 = {
|
|
178
|
+
id: "MM003",
|
|
179
|
+
severity: "high",
|
|
180
|
+
title: "volatile content in the always-loaded prefix",
|
|
181
|
+
run({ workspace }) {
|
|
182
|
+
const out = [];
|
|
183
|
+
for (const file of alwaysLoaded(workspace)) {
|
|
184
|
+
// The generated OnDemandMemory list is a routing manifest, not prose: its labels are not status.
|
|
185
|
+
const lines = stripGeneratedRegions(file.lines);
|
|
186
|
+
for (const s of sections(lines)) {
|
|
187
|
+
const sectionLines = lines.slice(s.startLine - 1, s.endLine);
|
|
188
|
+
const reasons = sectionVolatilityReasons(sectionLines);
|
|
189
|
+
if (reasons.length === 0)
|
|
190
|
+
continue;
|
|
191
|
+
const text = sectionLines.join("\n");
|
|
192
|
+
out.push({
|
|
193
|
+
rule: "MM003",
|
|
194
|
+
severity: "high",
|
|
195
|
+
message: `"${s.heading}" (lines ${s.startLine}-${s.endLine}) holds ${reasons.join(" and ")} in the always-loaded prefix`,
|
|
196
|
+
file: file.rel,
|
|
197
|
+
startLine: s.startLine,
|
|
198
|
+
endLine: s.endLine,
|
|
199
|
+
tokens: estimateTokens(text),
|
|
200
|
+
hint: "move it to an OnDemandMemory file; editing it here invalidates the prompt cache every time",
|
|
201
|
+
});
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
return out.sort((a, b) => (b.tokens ?? 0) - (a.tokens ?? 0));
|
|
205
|
+
},
|
|
206
|
+
};
|
|
207
|
+
// ---------------------------------------------------------------------------
|
|
208
|
+
// MM004 an OnDemandMemory file exists but is unreferenced by any OnDemandMemory list
|
|
209
|
+
// ---------------------------------------------------------------------------
|
|
210
|
+
/**
|
|
211
|
+
* Does `source` name one of `candidates` in a way an agent could actually open: by relative
|
|
212
|
+
* path, or by filename including its extension.
|
|
213
|
+
*
|
|
214
|
+
* The bare filename stem does NOT count. It used to, and it made the rule near-useless on real
|
|
215
|
+
* corpora: an OnDemandMemory file called `safety.md` was treated as routed by any file
|
|
216
|
+
* containing the word "safety" in ordinary prose, so an orphan went unreported the moment its
|
|
217
|
+
* topic was mentioned anywhere. That is the same false positive the 2026-09-04 list rewrite hit
|
|
218
|
+
* from the other side, when list prose containing the word "core" was read as a link to core.md.
|
|
219
|
+
*
|
|
220
|
+
* One pass over `source.content` per source file, not one `.includes()` scan per candidate:
|
|
221
|
+
* builds a single regex alternation of every candidate's rel path and basename, escaped, then
|
|
222
|
+
* matches it once. The original pairwise version (`source.content.includes(target.rel) ||
|
|
223
|
+
* source.content.includes(base)`, called once per source/candidate pair) measured MM004 alone at
|
|
224
|
+
* ~320ms wall time against a real 68-file personal memory directory, over the 250ms bar this
|
|
225
|
+
* rewrite was measured against (2026-09-13 audit round 2, D9).
|
|
226
|
+
*/
|
|
227
|
+
function referencedIn(source, candidates) {
|
|
228
|
+
const escapeRegExp = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
229
|
+
const byName = new Map();
|
|
230
|
+
for (const c of candidates) {
|
|
231
|
+
for (const name of new Set([c.rel, path.basename(c.rel)])) {
|
|
232
|
+
const list = byName.get(name);
|
|
233
|
+
if (list)
|
|
234
|
+
list.push(c);
|
|
235
|
+
else
|
|
236
|
+
byName.set(name, [c]);
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
const found = new Set();
|
|
240
|
+
if (byName.size === 0)
|
|
241
|
+
return found;
|
|
242
|
+
const pattern = new RegExp([...byName.keys()].map(escapeRegExp).join("|"), "g");
|
|
243
|
+
for (const match of source.content.matchAll(pattern)) {
|
|
244
|
+
for (const c of byName.get(match[0]) ?? [])
|
|
245
|
+
found.add(c);
|
|
246
|
+
}
|
|
247
|
+
return found;
|
|
248
|
+
}
|
|
249
|
+
/**
|
|
250
|
+
* Auto-memory topic files and compiled OnDemandMemory files are routed by different lists, so a
|
|
251
|
+
* reference only counts inside its own routing domain.
|
|
252
|
+
*/
|
|
253
|
+
function sameDomain(a, b) {
|
|
254
|
+
return a.rel.startsWith("(auto-memory)/") === b.rel.startsWith("(auto-memory)/");
|
|
255
|
+
}
|
|
256
|
+
/**
|
|
257
|
+
* Every file reachable from `list`, following references transitively.
|
|
258
|
+
*
|
|
259
|
+
* Reachability, not list membership, is what routing actually requires. A file the list names
|
|
260
|
+
* directly is reachable in one hop; a file named only by a file the list names is reachable in
|
|
261
|
+
* two, and the agent gets there just as reliably. Splitting an append-only changelog out of a
|
|
262
|
+
* topic file and leaving a pointer behind produces exactly that shape, and treating the result as
|
|
263
|
+
* orphaned would make the rule fire on a correct reorganization.
|
|
264
|
+
*
|
|
265
|
+
* The reference pairs are computed once, in a single pass, before the walk: doing `includes`
|
|
266
|
+
* inside the fixpoint loop instead is quadratic over file contents on every round.
|
|
267
|
+
*/
|
|
268
|
+
function reachableFrom(list, files) {
|
|
269
|
+
const candidates = files.filter((f) => f !== list && sameDomain(f, list));
|
|
270
|
+
const sources = [list, ...candidates];
|
|
271
|
+
const edges = new Map();
|
|
272
|
+
for (const source of sources) {
|
|
273
|
+
const targets = candidates.filter((target) => target !== source);
|
|
274
|
+
edges.set(source, [...referencedIn(source, targets)]);
|
|
275
|
+
}
|
|
276
|
+
const reached = new Set([list]);
|
|
277
|
+
const queue = [list];
|
|
278
|
+
while (queue.length > 0) {
|
|
279
|
+
const current = queue.shift();
|
|
280
|
+
for (const next of edges.get(current) ?? []) {
|
|
281
|
+
if (reached.has(next))
|
|
282
|
+
continue;
|
|
283
|
+
reached.add(next);
|
|
284
|
+
queue.push(next);
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
return reached;
|
|
288
|
+
}
|
|
289
|
+
const mm004 = {
|
|
290
|
+
id: "MM004",
|
|
291
|
+
severity: "low",
|
|
292
|
+
title: "OnDemandMemory file unreachable from any OnDemandMemory list",
|
|
293
|
+
run({ workspace }) {
|
|
294
|
+
const out = [];
|
|
295
|
+
const cache = new Map();
|
|
296
|
+
for (const file of workspace.files) {
|
|
297
|
+
// Each on-demand file is checked against the OnDemandMemory list that routes to it:
|
|
298
|
+
// auto-memory topic files against the auto-memory MEMORY.md, compiled OnDemandMemory
|
|
299
|
+
// files against index.md.
|
|
300
|
+
const list = file.rel.startsWith("(auto-memory)/") ? workspace.autoMemoryList : workspace.onDemandList;
|
|
301
|
+
if (!list || file === list || file.alwaysLoaded)
|
|
302
|
+
continue;
|
|
303
|
+
// AlwaysOnMemory.md's alwaysOn/onDemandList views are the prefix, not routed content: when
|
|
304
|
+
// the stub inlines them they are not always-loaded as files, but nothing routes to them
|
|
305
|
+
// either. Until the 2026-09-04 list rewrite they passed only because the list prose
|
|
306
|
+
// happened to contain the word "core", which referencesFile() took as a link.
|
|
307
|
+
if (file.generated || file.kind === "alwaysOn" || file.kind === "onDemandList")
|
|
308
|
+
continue;
|
|
309
|
+
let reached = cache.get(list);
|
|
310
|
+
if (!reached) {
|
|
311
|
+
reached = reachableFrom(list, workspace.files);
|
|
312
|
+
cache.set(list, reached);
|
|
313
|
+
}
|
|
314
|
+
if (reached.has(file))
|
|
315
|
+
continue;
|
|
316
|
+
out.push({
|
|
317
|
+
rule: "MM004",
|
|
318
|
+
severity: "low",
|
|
319
|
+
message: `${file.rel} is not reachable from ${list.rel}`,
|
|
320
|
+
file: file.rel,
|
|
321
|
+
tokens: file.tokens,
|
|
322
|
+
hint: "reference it from the OnDemandMemory list, or from a file the list already routes to; unroutable content is invisible to the agent",
|
|
323
|
+
});
|
|
324
|
+
}
|
|
325
|
+
return out.sort((a, b) => (b.tokens ?? 0) - (a.tokens ?? 0));
|
|
326
|
+
},
|
|
327
|
+
};
|
|
328
|
+
// ---------------------------------------------------------------------------
|
|
329
|
+
// MM005 substantially duplicated content
|
|
330
|
+
// ---------------------------------------------------------------------------
|
|
331
|
+
const MIN_DUPLICATE_CHARS = 80;
|
|
332
|
+
/**
|
|
333
|
+
* Blocks repeated verbatim (modulo list-marker/heading-hash/whitespace normalisation) across two
|
|
334
|
+
* or more files. Shared by MM005 and the multi-file scan (scan.ts) so both see the same picture
|
|
335
|
+
* of what a reorganize plan should merge.
|
|
336
|
+
*/
|
|
337
|
+
export function findDuplicateBlocks(files) {
|
|
338
|
+
const seen = new Map();
|
|
339
|
+
for (const file of files) {
|
|
340
|
+
// A generated mirror duplicates its stub on purpose.
|
|
341
|
+
if (file.generated)
|
|
342
|
+
continue;
|
|
343
|
+
for (const block of fileBlocks(file)) {
|
|
344
|
+
const key = normalise(block.text);
|
|
345
|
+
if (key.length < MIN_DUPLICATE_CHARS)
|
|
346
|
+
continue;
|
|
347
|
+
const list = seen.get(key) ?? [];
|
|
348
|
+
list.push({ file, block });
|
|
349
|
+
seen.set(key, list);
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
const out = [];
|
|
353
|
+
for (const occurrences of seen.values()) {
|
|
354
|
+
if (occurrences.length < 2)
|
|
355
|
+
continue;
|
|
356
|
+
const first = occurrences[0];
|
|
357
|
+
if (!first)
|
|
358
|
+
continue;
|
|
359
|
+
const wastedTokens = estimateTokens(first.block.text) * (occurrences.length - 1);
|
|
360
|
+
out.push({ occurrences, wastedTokens });
|
|
361
|
+
}
|
|
362
|
+
return out.sort((a, b) => b.wastedTokens - a.wastedTokens);
|
|
363
|
+
}
|
|
364
|
+
const mm005 = {
|
|
365
|
+
id: "MM005",
|
|
366
|
+
severity: "med",
|
|
367
|
+
title: "duplicated content across memory files",
|
|
368
|
+
run({ workspace }) {
|
|
369
|
+
const groups = findDuplicateBlocks(workspace.files);
|
|
370
|
+
const out = groups.map(({ occurrences, wastedTokens }) => {
|
|
371
|
+
const first = occurrences[0];
|
|
372
|
+
const where = occurrences
|
|
373
|
+
.slice(1)
|
|
374
|
+
.map((o) => `${o.file.rel}:${o.block.startLine}`)
|
|
375
|
+
.join(", ");
|
|
376
|
+
return {
|
|
377
|
+
rule: "MM005",
|
|
378
|
+
severity: "med",
|
|
379
|
+
message: `this block appears ${occurrences.length} times, also at ${where}`,
|
|
380
|
+
file: first.file.rel,
|
|
381
|
+
startLine: first.block.startLine,
|
|
382
|
+
endLine: first.block.endLine,
|
|
383
|
+
tokens: wastedTokens,
|
|
384
|
+
hint: "keep one copy and reference it; duplicates drift apart over time",
|
|
385
|
+
};
|
|
386
|
+
});
|
|
387
|
+
return out;
|
|
388
|
+
},
|
|
389
|
+
};
|
|
390
|
+
// ---------------------------------------------------------------------------
|
|
391
|
+
// MM006 OnDemandMemory list line exceeds the character cap
|
|
392
|
+
// ---------------------------------------------------------------------------
|
|
393
|
+
/**
|
|
394
|
+
* The hook is the descriptive text, with the list marker, any leading markdown link, and a
|
|
395
|
+
* separator dash stripped. That is the part a human writes and the part worth budgeting;
|
|
396
|
+
* the link is structural overhead the author cannot shorten much.
|
|
397
|
+
*/
|
|
398
|
+
export function hookText(line) {
|
|
399
|
+
return line
|
|
400
|
+
.replace(/^\s*[-*+]\s+/, "")
|
|
401
|
+
.replace(/^\[[^\]]*\]\([^)]*\)\s*/, "")
|
|
402
|
+
.replace(/^[—–:-]\s*/, "")
|
|
403
|
+
.trim();
|
|
404
|
+
}
|
|
405
|
+
const mm006 = {
|
|
406
|
+
id: "MM006",
|
|
407
|
+
severity: "low",
|
|
408
|
+
title: "OnDemandMemory list line exceeds the character cap",
|
|
409
|
+
run({ workspace, config }) {
|
|
410
|
+
const lists = [workspace.onDemandList, workspace.autoMemoryList].filter((f, i, all) => f !== undefined && all.indexOf(f) === i);
|
|
411
|
+
const out = [];
|
|
412
|
+
for (const list of lists) {
|
|
413
|
+
list.lines.forEach((line, i) => {
|
|
414
|
+
if (!/^\s*[-*+]\s+/.test(line))
|
|
415
|
+
return;
|
|
416
|
+
const hook = hookText(line);
|
|
417
|
+
if (hook.length <= config.maxListLine)
|
|
418
|
+
return;
|
|
419
|
+
out.push({
|
|
420
|
+
rule: "MM006",
|
|
421
|
+
severity: "low",
|
|
422
|
+
message: `OnDemandMemory list hook is ${hook.length} characters, over the ${config.maxListLine} cap (line is ${line.length})`,
|
|
423
|
+
file: list.rel,
|
|
424
|
+
startLine: i + 1,
|
|
425
|
+
endLine: i + 1,
|
|
426
|
+
hint: "an OnDemandMemory list line is a hook, not a summary, and it is charged every turn",
|
|
427
|
+
});
|
|
428
|
+
});
|
|
429
|
+
}
|
|
430
|
+
return out;
|
|
431
|
+
},
|
|
432
|
+
};
|
|
433
|
+
// ---------------------------------------------------------------------------
|
|
434
|
+
// MM007 flat memory directory with no OnDemandMemory list
|
|
435
|
+
// ---------------------------------------------------------------------------
|
|
436
|
+
const mm007 = {
|
|
437
|
+
id: "MM007",
|
|
438
|
+
severity: "med",
|
|
439
|
+
title: "flat memory directory with no OnDemandMemory list",
|
|
440
|
+
run({ workspace }) {
|
|
441
|
+
if ((workspace.shape !== "memory-dir" && workspace.shape !== "auto-memory") || workspace.onDemandList)
|
|
442
|
+
return [];
|
|
443
|
+
if (workspace.files.length < 5)
|
|
444
|
+
return [];
|
|
445
|
+
const total = workspace.files.reduce((n, f) => n + f.tokens, 0);
|
|
446
|
+
return [
|
|
447
|
+
{
|
|
448
|
+
rule: "MM007",
|
|
449
|
+
severity: "med",
|
|
450
|
+
message: `${workspace.files.length} memory files with no OnDemandMemory list, so all ${formatTokens(total)} tokens get read`,
|
|
451
|
+
file: ".",
|
|
452
|
+
tokens: total,
|
|
453
|
+
hint: "add an OnDemandMemory list with one routing line per file so the agent can load on demand",
|
|
454
|
+
},
|
|
455
|
+
];
|
|
456
|
+
},
|
|
457
|
+
};
|
|
458
|
+
// ---------------------------------------------------------------------------
|
|
459
|
+
// MM008 credential-shaped string in a memory file
|
|
460
|
+
// ---------------------------------------------------------------------------
|
|
461
|
+
const SECRETS = [
|
|
462
|
+
{ re: /\bsk-ant-[A-Za-z0-9_-]{16,}/g, what: "an Anthropic API key" },
|
|
463
|
+
{ re: /\bsk-[A-Za-z0-9]{32,}/g, what: "an OpenAI-style API key" },
|
|
464
|
+
{ re: /\bnvapi-[A-Za-z0-9_-]{16,}/g, what: "an NVIDIA API key" },
|
|
465
|
+
{ re: /\b(?:ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9]{20,}/g, what: "a GitHub token" },
|
|
466
|
+
{ re: /\bgithub_pat_[A-Za-z0-9_]{20,}/g, what: "a GitHub fine-grained token" },
|
|
467
|
+
{ re: /\bAKIA[0-9A-Z]{16}\b/g, what: "an AWS access key id" },
|
|
468
|
+
{ re: /\bAIza[A-Za-z0-9_-]{35}\b/g, what: "a Google API key" },
|
|
469
|
+
{ re: /\bxox[abprs]-[A-Za-z0-9-]{10,}/g, what: "a Slack token" },
|
|
470
|
+
// Stripe: underscore-separated, so the OpenAI-style `sk-` pattern above never matched these.
|
|
471
|
+
{ re: /\b(?:sk|rk)_(?:live|test)_[A-Za-z0-9]{16,}/g, what: "a Stripe secret key" },
|
|
472
|
+
{ re: /\bwhsec_[A-Za-z0-9]{16,}/g, what: "a Stripe webhook secret" },
|
|
473
|
+
{ re: /-----BEGIN (?:RSA |EC |OPENSSH |PGP )?PRIVATE KEY-----/g, what: "a private key block" },
|
|
474
|
+
{ re: /\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\b/g, what: "a JSON Web Token" },
|
|
475
|
+
{ re: /\brnd_[A-Za-z0-9]{20,}\b/g, what: "a Render API key" },
|
|
476
|
+
];
|
|
477
|
+
/** Never print a credential back out. Show just enough to locate it. */
|
|
478
|
+
function redact(match) {
|
|
479
|
+
const head = match.slice(0, Math.min(8, match.length));
|
|
480
|
+
return `${head}${"*".repeat(6)}`;
|
|
481
|
+
}
|
|
482
|
+
/** MM008 for one file. Exported so `init` can refuse to copy a credential into new files. */
|
|
483
|
+
export function secretFindings(file) {
|
|
484
|
+
const out = [];
|
|
485
|
+
file.lines.forEach((line, i) => {
|
|
486
|
+
for (const { re, what } of SECRETS) {
|
|
487
|
+
re.lastIndex = 0;
|
|
488
|
+
// Every match on the line, not just the first: two keys assigned on one line are two
|
|
489
|
+
// credentials to rotate, and reporting one of them reads as though the other is fine.
|
|
490
|
+
let m;
|
|
491
|
+
while ((m = re.exec(line)) !== null) {
|
|
492
|
+
if (!m[0])
|
|
493
|
+
break;
|
|
494
|
+
out.push({
|
|
495
|
+
rule: "MM008",
|
|
496
|
+
severity: "high",
|
|
497
|
+
message: `looks like ${what}: ${redact(m[0])}`,
|
|
498
|
+
file: file.rel,
|
|
499
|
+
startLine: i + 1,
|
|
500
|
+
endLine: i + 1,
|
|
501
|
+
hint: "memory files get committed and pasted into prompts; rotate it and remove it",
|
|
502
|
+
});
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
});
|
|
506
|
+
return out;
|
|
507
|
+
}
|
|
508
|
+
const mm008 = {
|
|
509
|
+
id: "MM008",
|
|
510
|
+
severity: "high",
|
|
511
|
+
title: "credential-shaped string in a memory file",
|
|
512
|
+
run({ workspace }) {
|
|
513
|
+
return workspace.files.flatMap(secretFindings);
|
|
514
|
+
},
|
|
515
|
+
};
|
|
516
|
+
// ---------------------------------------------------------------------------
|
|
517
|
+
// ---------------------------------------------------------------------------
|
|
518
|
+
// MM009 a flat section over the granularity threshold in a long-term file
|
|
519
|
+
// ---------------------------------------------------------------------------
|
|
520
|
+
export { GRANULARITY_TOKENS };
|
|
521
|
+
/**
|
|
522
|
+
* sections() yields leaves: a section ends where the next heading of any level starts, so a
|
|
523
|
+
* leaf this large has no subheading by construction. An episodic section (mostly dated
|
|
524
|
+
* bullets) is exempt because the router already addresses it one entry at a time (O4).
|
|
525
|
+
* The 2026-09-04 audit found a 4,467-token "Current state" with no subheading; any recall
|
|
526
|
+
* that touched it paid all of it.
|
|
527
|
+
*/
|
|
528
|
+
const mm009 = {
|
|
529
|
+
id: "MM009",
|
|
530
|
+
severity: "low",
|
|
531
|
+
title: "flat section over the granularity threshold",
|
|
532
|
+
run({ workspace }) {
|
|
533
|
+
const out = [];
|
|
534
|
+
for (const file of workspace.files) {
|
|
535
|
+
if (file.alwaysLoaded || file === workspace.onDemandList || file === workspace.autoMemoryList)
|
|
536
|
+
continue;
|
|
537
|
+
if (file.generated || file.kind === "alwaysOn" || file.kind === "onDemandList")
|
|
538
|
+
continue;
|
|
539
|
+
for (const s of sections(file.lines)) {
|
|
540
|
+
if (s.level === 0)
|
|
541
|
+
continue;
|
|
542
|
+
const tokens = estimateTokens(file.lines.slice(s.startLine - 1, s.endLine).join("\n"));
|
|
543
|
+
if (tokens < GRANULARITY_TOKENS)
|
|
544
|
+
continue;
|
|
545
|
+
const body = file.lines.slice(s.startLine, s.endLine);
|
|
546
|
+
if (isEpisodicBody(body))
|
|
547
|
+
continue;
|
|
548
|
+
out.push({
|
|
549
|
+
rule: "MM009",
|
|
550
|
+
severity: "low",
|
|
551
|
+
message: `"${s.heading}" is ${formatTokens(tokens)} tokens with no subheading, so a recall that touches it pays all of it`,
|
|
552
|
+
file: file.rel,
|
|
553
|
+
startLine: s.startLine,
|
|
554
|
+
endLine: s.endLine,
|
|
555
|
+
tokens,
|
|
556
|
+
hint: "add H3 subheadings so retrieval can return the part the task needs (O7)",
|
|
557
|
+
});
|
|
558
|
+
}
|
|
559
|
+
}
|
|
560
|
+
return out.sort((a, b) => (b.tokens ?? 0) - (a.tokens ?? 0));
|
|
561
|
+
},
|
|
562
|
+
};
|
|
563
|
+
// ---------------------------------------------------------------------------
|
|
564
|
+
// MM010 a compiled workspace has drifted from what init wrote
|
|
565
|
+
// ---------------------------------------------------------------------------
|
|
566
|
+
/** CRLF-normalised, because a checkout can rewrite line endings without anyone editing. */
|
|
567
|
+
function normalised(content) {
|
|
568
|
+
return content.replace(/\r\n/g, "\n");
|
|
569
|
+
}
|
|
570
|
+
/**
|
|
571
|
+
* The re-apply case: someone comes back to a memory they optimised earlier. `doctor` is the
|
|
572
|
+
* front door for that, so it compares every OnDemandMemory file, the AlwaysOnMemory body and the
|
|
573
|
+
* OnDemandMemory list on disk against the hashes `init` recorded in manifest.json and names each
|
|
574
|
+
* file that moved, was removed, or appeared. The remedy is always the same and is printed with
|
|
575
|
+
* every finding. Read-only, as every rule is; the writing hand stays `init --update`.
|
|
576
|
+
*
|
|
577
|
+
* OnDemandMemory file hashes were taken over the trimmed content and the file is written with
|
|
578
|
+
* one trailing newline, so the OnDemandMemory file side compares trimmed; AlwaysOnMemory.md was
|
|
579
|
+
* written verbatim.
|
|
580
|
+
*/
|
|
581
|
+
const mm010 = {
|
|
582
|
+
id: "MM010",
|
|
583
|
+
severity: "med",
|
|
584
|
+
title: "compiled workspace has drifted since init",
|
|
585
|
+
run({ workspace }) {
|
|
586
|
+
const manifest = workspace.manifest;
|
|
587
|
+
if (workspace.shape !== "compiled" || !manifest)
|
|
588
|
+
return [];
|
|
589
|
+
const out = [];
|
|
590
|
+
const remedy = "re-apply with: minnimemory init --update (keeps every OnDemandMemory file, rewrites the OnDemandMemory list, manifest and stub)";
|
|
591
|
+
const byRel = new Map(workspace.files.map((f) => [f.rel.replace(/\\/g, "/"), f]));
|
|
592
|
+
const compiledDir = COMPILED_DIR_NAME;
|
|
593
|
+
const seen = new Set();
|
|
594
|
+
for (const m of manifest.onDemandFiles) {
|
|
595
|
+
const rel = `${compiledDir}/${m.file}`;
|
|
596
|
+
seen.add(rel);
|
|
597
|
+
const file = byRel.get(rel);
|
|
598
|
+
if (!file) {
|
|
599
|
+
out.push({ rule: "MM010", severity: "med", message: `${rel} is in the manifest but missing on disk`, file: rel, hint: remedy });
|
|
600
|
+
continue;
|
|
601
|
+
}
|
|
602
|
+
if (driftHash(file.content) !== m.hash) {
|
|
603
|
+
out.push({
|
|
604
|
+
rule: "MM010",
|
|
605
|
+
severity: "med",
|
|
606
|
+
message: `${rel} was edited since init wrote it`,
|
|
607
|
+
file: rel,
|
|
608
|
+
tokens: file.tokens,
|
|
609
|
+
hint: remedy,
|
|
610
|
+
});
|
|
611
|
+
}
|
|
612
|
+
}
|
|
613
|
+
for (const f of workspace.files) {
|
|
614
|
+
const rel = f.rel.replace(/\\/g, "/");
|
|
615
|
+
if (f.kind === "onDemand" && !seen.has(rel)) {
|
|
616
|
+
out.push({ rule: "MM010", severity: "med", message: `${rel} is on disk but not in the manifest`, file: rel, tokens: f.tokens, hint: remedy });
|
|
617
|
+
}
|
|
618
|
+
}
|
|
619
|
+
{
|
|
620
|
+
const entry = manifest.always;
|
|
621
|
+
const rel = `${compiledDir}/${entry.file}`;
|
|
622
|
+
const file = byRel.get(rel);
|
|
623
|
+
if (file && verbatimHash(normalised(file.content)) !== entry.hash) {
|
|
624
|
+
out.push({ rule: "MM010", severity: "med", message: `${rel} was edited since init wrote it`, file: rel, tokens: file.tokens, hint: remedy });
|
|
625
|
+
}
|
|
626
|
+
}
|
|
627
|
+
// The stub is the always-loaded prefix, so content appended to it is the drift that costs
|
|
628
|
+
// most: it is paid on every turn and routed by nothing. Without this check doctor reported
|
|
629
|
+
// "no drift since init. Nothing to re-apply." with un-routed content sitting in the host
|
|
630
|
+
// file, and MM001 only noticed once the regrown prefix crossed the budget.
|
|
631
|
+
if (manifest.stub) {
|
|
632
|
+
const rel = manifest.stub.file;
|
|
633
|
+
const file = byRel.get(rel);
|
|
634
|
+
if (file && verbatimHash(normalised(file.content)) !== manifest.stub.hash) {
|
|
635
|
+
out.push({
|
|
636
|
+
rule: "MM010",
|
|
637
|
+
severity: "med",
|
|
638
|
+
message: `${rel} was edited since init wrote it`,
|
|
639
|
+
file: rel,
|
|
640
|
+
tokens: file.tokens,
|
|
641
|
+
hint: remedy,
|
|
642
|
+
});
|
|
643
|
+
}
|
|
644
|
+
}
|
|
645
|
+
return out;
|
|
646
|
+
},
|
|
647
|
+
};
|
|
648
|
+
export const RULES = [mm001, mm002, mm003, mm004, mm005, mm006, mm007, mm008, mm009, mm010];
|
|
649
|
+
export function ruleById(id) {
|
|
650
|
+
return RULES.find((r) => r.id.toLowerCase() === id.toLowerCase());
|
|
651
|
+
}
|