minnimemory 1.0.0-beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/LICENSE +39 -0
  2. package/README.md +824 -0
  3. package/dist/bench.d.ts +98 -0
  4. package/dist/bench.js +142 -0
  5. package/dist/benchReport.d.ts +12 -0
  6. package/dist/benchReport.js +128 -0
  7. package/dist/bounds.d.ts +40 -0
  8. package/dist/bounds.js +44 -0
  9. package/dist/cli.d.ts +15 -0
  10. package/dist/cli.js +503 -0
  11. package/dist/compile.d.ts +187 -0
  12. package/dist/compile.js +516 -0
  13. package/dist/discover.d.ts +125 -0
  14. package/dist/discover.js +520 -0
  15. package/dist/doctor.d.ts +9 -0
  16. package/dist/doctor.js +67 -0
  17. package/dist/episodic.d.ts +47 -0
  18. package/dist/episodic.js +130 -0
  19. package/dist/hook.d.ts +45 -0
  20. package/dist/hook.js +104 -0
  21. package/dist/index.d.ts +18 -0
  22. package/dist/index.js +18 -0
  23. package/dist/init.d.ts +125 -0
  24. package/dist/init.js +475 -0
  25. package/dist/instructions.d.ts +60 -0
  26. package/dist/instructions.js +270 -0
  27. package/dist/mcp.d.ts +109 -0
  28. package/dist/mcp.js +252 -0
  29. package/dist/mcpServer.d.ts +136 -0
  30. package/dist/mcpServer.js +997 -0
  31. package/dist/paths.d.ts +25 -0
  32. package/dist/paths.js +47 -0
  33. package/dist/recall.d.ts +113 -0
  34. package/dist/recall.js +256 -0
  35. package/dist/recallDir.d.ts +50 -0
  36. package/dist/recallDir.js +187 -0
  37. package/dist/reorganize.d.ts +62 -0
  38. package/dist/reorganize.js +216 -0
  39. package/dist/report.d.ts +16 -0
  40. package/dist/report.js +204 -0
  41. package/dist/router.d.ts +141 -0
  42. package/dist/router.js +314 -0
  43. package/dist/rules.d.ts +32 -0
  44. package/dist/rules.js +651 -0
  45. package/dist/scan.d.ts +110 -0
  46. package/dist/scan.js +173 -0
  47. package/dist/text.d.ts +158 -0
  48. package/dist/text.js +395 -0
  49. package/dist/tokenizer.d.ts +26 -0
  50. package/dist/tokenizer.js +69 -0
  51. package/dist/types.d.ts +156 -0
  52. package/dist/types.js +17 -0
  53. package/dist/version.d.ts +7 -0
  54. package/dist/version.js +7 -0
  55. package/dist/writeProtocol.d.ts +19 -0
  56. package/dist/writeProtocol.js +45 -0
  57. package/examples/CLAUDE.md +75 -0
  58. package/examples/README.md +7 -0
  59. package/package.json +52 -0
package/dist/rules.js ADDED
@@ -0,0 +1,651 @@
1
+ /**
2
+ * The MM001-MM010 rule set.
3
+ *
4
+ * Every rule is pure: it reads a Workspace and returns Findings. Nothing here touches the
5
+ * filesystem or the network, which is what makes the whole engine testable from fixtures.
6
+ */
7
+ import fs from "node:fs";
8
+ import path from "node:path";
9
+ import { COMPILED_DIR_NAME, driftHash, verbatimHash } from "./compile.js";
10
+ import { blocks, GRANULARITY_TOKENS, isEpisodicBody, normalise, sections, sectionVolatilityReasons, stripGeneratedRegions, } from "./text.js";
11
+ import { estimateTokens, formatTokens } from "./tokenizer.js";
12
+ import { alwaysLoaded } from "./types.js";
13
+ // ---------------------------------------------------------------------------
14
+ // shared helpers
15
+ // ---------------------------------------------------------------------------
16
+ /** Sections of a memory file, adapted to the MemoryFile wrapper. */
17
+ function fileSections(file) {
18
+ return sections(file.lines);
19
+ }
20
+ function sectionText(file, s) {
21
+ return file.lines.slice(s.startLine - 1, s.endLine).join("\n");
22
+ }
23
+ function fileBlocks(file) {
24
+ return blocks(file.lines);
25
+ }
26
+ /**
27
+ * Rules return every finding they have. Truncation is a display concern and lives in the human
28
+ * renderer (report.ts), so `--json` and the exit code always see the complete list: capping here
29
+ * made the overflow marker's own "re-run with --json for the full list" hint false, and put the
30
+ * findings it hid out of reach of every output mode.
31
+ */
32
+ // ---------------------------------------------------------------------------
33
+ // MM001 always-loaded memory exceeds the token budget
34
+ // ---------------------------------------------------------------------------
35
+ const mm001 = {
36
+ id: "MM001",
37
+ severity: "high",
38
+ title: "always-loaded memory exceeds the token budget",
39
+ run({ workspace, config }) {
40
+ const loaded = alwaysLoaded(workspace);
41
+ const total = loaded.reduce((n, f) => n + f.tokens, 0);
42
+ if (total <= config.budget)
43
+ return [];
44
+ const biggest = [...loaded].sort((a, b) => b.tokens - a.tokens)[0];
45
+ const where = loaded.length === 1
46
+ ? `${biggest?.rel} is ${formatTokens(total)} tokens`
47
+ : `${loaded.length} always-loaded files total ${formatTokens(total)} tokens`;
48
+ return [
49
+ {
50
+ rule: "MM001",
51
+ severity: "high",
52
+ message: `${where}, resent every turn`,
53
+ file: biggest?.rel ?? ".",
54
+ tokens: total,
55
+ hint: `budget is ${formatTokens(config.budget)}; move task-specific content into routed OnDemandMemory files`,
56
+ },
57
+ ];
58
+ },
59
+ };
60
+ // ---------------------------------------------------------------------------
61
+ // MM002 content derivable from the repo itself
62
+ // ---------------------------------------------------------------------------
63
+ const TREE_CHAR = /[│├└┌┐┘┬┴┼]/;
64
+ /**
65
+ * One line of a directory listing, in any of the shapes real memory files use:
66
+ * a bare path, a tree-drawn path, either of those with a trailing "# comment" or "- comment",
67
+ * or a table row whose first cell is a path. Found by the 2026-09-02 sweep: the most common
68
+ * real-world shape is a Markdown table of directories, which the old bare-path pattern missed.
69
+ */
70
+ const PATH_TOKEN = /`?([\w.@+\-<>]+(?:[/\\][\w.@+\-<>*]*)*[/\\]?)`?/;
71
+ // No `^\s*` before a class that also contains `\s`: two adjacent whitespace quantifiers made the
72
+ // old PATHY pattern quadratic on a line of spaces (security audit 2026-09-02, finding 3).
73
+ const TREE_LINE = new RegExp(`^[│├└─|\\s]*${PATH_TOKEN.source}\\s*(?:(?:#|//|-|—|:)\\s.*)?$`);
74
+ const TABLE_ROW = new RegExp(`^\\s*\\|\\s*${PATH_TOKEN.source}\\s*\\|`);
75
+ /** Does the path token look like a path rather than a plain word: a slash, or a file extension. */
76
+ function pathLike(token) {
77
+ return /[/\\]/.test(token) || /\.[a-z]{1,5}$/i.test(token);
78
+ }
79
+ function treeRuns(file) {
80
+ const out = [];
81
+ let start = -1;
82
+ let treeChars = 0;
83
+ let paths = 0;
84
+ let tableRows = 0;
85
+ const close = (end) => {
86
+ if (start >= 0) {
87
+ const len = end - start + 1;
88
+ if (len >= 6 && (treeChars > 0 || paths >= 3)) {
89
+ const text = file.lines.slice(start - 1, end).join("\n");
90
+ const what = tableRows > len / 2 ? "a directory listing" : "a directory tree";
91
+ out.push({
92
+ rule: "MM002",
93
+ severity: "med",
94
+ message: `lines ${start}-${end} are ${what} derivable from the repo itself`,
95
+ file: file.rel,
96
+ startLine: start,
97
+ endLine: end,
98
+ tokens: estimateTokens(text),
99
+ hint: "delete it; the agent can list the directory, and this copy goes stale silently",
100
+ });
101
+ }
102
+ }
103
+ start = -1;
104
+ treeChars = 0;
105
+ paths = 0;
106
+ tableRows = 0;
107
+ };
108
+ file.lines.forEach((line, i) => {
109
+ const n = i + 1;
110
+ if (line.trim() === "")
111
+ return close(i);
112
+ const row = TABLE_ROW.exec(line);
113
+ const tree = row ? null : TREE_LINE.exec(line);
114
+ const token = row?.[1] ?? tree?.[1];
115
+ const isTree = tree !== null && tree !== undefined && TREE_CHAR.test(line);
116
+ if (token && (isTree || pathLike(token))) {
117
+ if (start < 0)
118
+ start = n;
119
+ if (isTree)
120
+ treeChars++;
121
+ if (pathLike(token))
122
+ paths++;
123
+ if (row)
124
+ tableRows++;
125
+ }
126
+ else {
127
+ close(i);
128
+ }
129
+ });
130
+ close(file.lines.length);
131
+ return out;
132
+ }
133
+ function duplicatedScripts(file, root) {
134
+ const pkgPath = path.join(root, "package.json");
135
+ if (!fs.existsSync(pkgPath))
136
+ return [];
137
+ let names = [];
138
+ try {
139
+ const pkg = JSON.parse(fs.readFileSync(pkgPath, "utf8"));
140
+ names = Object.keys(pkg.scripts ?? {});
141
+ }
142
+ catch {
143
+ return [];
144
+ }
145
+ if (names.length === 0)
146
+ return [];
147
+ const hits = names.filter((n) => file.content.includes(`npm run ${n}`));
148
+ if (hits.length < 3)
149
+ return [];
150
+ return [
151
+ {
152
+ rule: "MM002",
153
+ severity: "med",
154
+ message: `${hits.length} npm scripts are restated here and already exist in package.json`,
155
+ file: file.rel,
156
+ hint: "point the agent at package.json instead of copying the script list",
157
+ },
158
+ ];
159
+ }
160
+ const mm002 = {
161
+ id: "MM002",
162
+ severity: "med",
163
+ title: "content derivable from the repo itself",
164
+ run({ workspace }) {
165
+ const out = [];
166
+ for (const file of workspace.files) {
167
+ out.push(...treeRuns(file));
168
+ if (file.alwaysLoaded)
169
+ out.push(...duplicatedScripts(file, workspace.root));
170
+ }
171
+ return out.sort((a, b) => (b.tokens ?? 0) - (a.tokens ?? 0));
172
+ },
173
+ };
174
+ // ---------------------------------------------------------------------------
175
+ // MM003 volatile content in the always-loaded prefix
176
+ // ---------------------------------------------------------------------------
177
+ const mm003 = {
178
+ id: "MM003",
179
+ severity: "high",
180
+ title: "volatile content in the always-loaded prefix",
181
+ run({ workspace }) {
182
+ const out = [];
183
+ for (const file of alwaysLoaded(workspace)) {
184
+ // The generated OnDemandMemory list is a routing manifest, not prose: its labels are not status.
185
+ const lines = stripGeneratedRegions(file.lines);
186
+ for (const s of sections(lines)) {
187
+ const sectionLines = lines.slice(s.startLine - 1, s.endLine);
188
+ const reasons = sectionVolatilityReasons(sectionLines);
189
+ if (reasons.length === 0)
190
+ continue;
191
+ const text = sectionLines.join("\n");
192
+ out.push({
193
+ rule: "MM003",
194
+ severity: "high",
195
+ message: `"${s.heading}" (lines ${s.startLine}-${s.endLine}) holds ${reasons.join(" and ")} in the always-loaded prefix`,
196
+ file: file.rel,
197
+ startLine: s.startLine,
198
+ endLine: s.endLine,
199
+ tokens: estimateTokens(text),
200
+ hint: "move it to an OnDemandMemory file; editing it here invalidates the prompt cache every time",
201
+ });
202
+ }
203
+ }
204
+ return out.sort((a, b) => (b.tokens ?? 0) - (a.tokens ?? 0));
205
+ },
206
+ };
207
+ // ---------------------------------------------------------------------------
208
+ // MM004 an OnDemandMemory file exists but is unreferenced by any OnDemandMemory list
209
+ // ---------------------------------------------------------------------------
210
+ /**
211
+ * Does `source` name one of `candidates` in a way an agent could actually open: by relative
212
+ * path, or by filename including its extension.
213
+ *
214
+ * The bare filename stem does NOT count. It used to, and it made the rule near-useless on real
215
+ * corpora: an OnDemandMemory file called `safety.md` was treated as routed by any file
216
+ * containing the word "safety" in ordinary prose, so an orphan went unreported the moment its
217
+ * topic was mentioned anywhere. That is the same false positive the 2026-09-04 list rewrite hit
218
+ * from the other side, when list prose containing the word "core" was read as a link to core.md.
219
+ *
220
+ * One pass over `source.content` per source file, not one `.includes()` scan per candidate:
221
+ * builds a single regex alternation of every candidate's rel path and basename, escaped, then
222
+ * matches it once. The original pairwise version (`source.content.includes(target.rel) ||
223
+ * source.content.includes(base)`, called once per source/candidate pair) measured MM004 alone at
224
+ * ~320ms wall time against a real 68-file personal memory directory, over the 250ms bar this
225
+ * rewrite was measured against (2026-09-13 audit round 2, D9).
226
+ */
227
+ function referencedIn(source, candidates) {
228
+ const escapeRegExp = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
229
+ const byName = new Map();
230
+ for (const c of candidates) {
231
+ for (const name of new Set([c.rel, path.basename(c.rel)])) {
232
+ const list = byName.get(name);
233
+ if (list)
234
+ list.push(c);
235
+ else
236
+ byName.set(name, [c]);
237
+ }
238
+ }
239
+ const found = new Set();
240
+ if (byName.size === 0)
241
+ return found;
242
+ const pattern = new RegExp([...byName.keys()].map(escapeRegExp).join("|"), "g");
243
+ for (const match of source.content.matchAll(pattern)) {
244
+ for (const c of byName.get(match[0]) ?? [])
245
+ found.add(c);
246
+ }
247
+ return found;
248
+ }
249
+ /**
250
+ * Auto-memory topic files and compiled OnDemandMemory files are routed by different lists, so a
251
+ * reference only counts inside its own routing domain.
252
+ */
253
+ function sameDomain(a, b) {
254
+ return a.rel.startsWith("(auto-memory)/") === b.rel.startsWith("(auto-memory)/");
255
+ }
256
+ /**
257
+ * Every file reachable from `list`, following references transitively.
258
+ *
259
+ * Reachability, not list membership, is what routing actually requires. A file the list names
260
+ * directly is reachable in one hop; a file named only by a file the list names is reachable in
261
+ * two, and the agent gets there just as reliably. Splitting an append-only changelog out of a
262
+ * topic file and leaving a pointer behind produces exactly that shape, and treating the result as
263
+ * orphaned would make the rule fire on a correct reorganization.
264
+ *
265
+ * The reference pairs are computed once, in a single pass, before the walk: doing `includes`
266
+ * inside the fixpoint loop instead is quadratic over file contents on every round.
267
+ */
268
+ function reachableFrom(list, files) {
269
+ const candidates = files.filter((f) => f !== list && sameDomain(f, list));
270
+ const sources = [list, ...candidates];
271
+ const edges = new Map();
272
+ for (const source of sources) {
273
+ const targets = candidates.filter((target) => target !== source);
274
+ edges.set(source, [...referencedIn(source, targets)]);
275
+ }
276
+ const reached = new Set([list]);
277
+ const queue = [list];
278
+ while (queue.length > 0) {
279
+ const current = queue.shift();
280
+ for (const next of edges.get(current) ?? []) {
281
+ if (reached.has(next))
282
+ continue;
283
+ reached.add(next);
284
+ queue.push(next);
285
+ }
286
+ }
287
+ return reached;
288
+ }
289
+ const mm004 = {
290
+ id: "MM004",
291
+ severity: "low",
292
+ title: "OnDemandMemory file unreachable from any OnDemandMemory list",
293
+ run({ workspace }) {
294
+ const out = [];
295
+ const cache = new Map();
296
+ for (const file of workspace.files) {
297
+ // Each on-demand file is checked against the OnDemandMemory list that routes to it:
298
+ // auto-memory topic files against the auto-memory MEMORY.md, compiled OnDemandMemory
299
+ // files against index.md.
300
+ const list = file.rel.startsWith("(auto-memory)/") ? workspace.autoMemoryList : workspace.onDemandList;
301
+ if (!list || file === list || file.alwaysLoaded)
302
+ continue;
303
+ // AlwaysOnMemory.md's alwaysOn/onDemandList views are the prefix, not routed content: when
304
+ // the stub inlines them they are not always-loaded as files, but nothing routes to them
305
+ // either. Until the 2026-09-04 list rewrite they passed only because the list prose
306
+ // happened to contain the word "core", which referencesFile() took as a link.
307
+ if (file.generated || file.kind === "alwaysOn" || file.kind === "onDemandList")
308
+ continue;
309
+ let reached = cache.get(list);
310
+ if (!reached) {
311
+ reached = reachableFrom(list, workspace.files);
312
+ cache.set(list, reached);
313
+ }
314
+ if (reached.has(file))
315
+ continue;
316
+ out.push({
317
+ rule: "MM004",
318
+ severity: "low",
319
+ message: `${file.rel} is not reachable from ${list.rel}`,
320
+ file: file.rel,
321
+ tokens: file.tokens,
322
+ hint: "reference it from the OnDemandMemory list, or from a file the list already routes to; unroutable content is invisible to the agent",
323
+ });
324
+ }
325
+ return out.sort((a, b) => (b.tokens ?? 0) - (a.tokens ?? 0));
326
+ },
327
+ };
328
+ // ---------------------------------------------------------------------------
329
+ // MM005 substantially duplicated content
330
+ // ---------------------------------------------------------------------------
331
+ const MIN_DUPLICATE_CHARS = 80;
332
+ /**
333
+ * Blocks repeated verbatim (modulo list-marker/heading-hash/whitespace normalisation) across two
334
+ * or more files. Shared by MM005 and the multi-file scan (scan.ts) so both see the same picture
335
+ * of what a reorganize plan should merge.
336
+ */
337
+ export function findDuplicateBlocks(files) {
338
+ const seen = new Map();
339
+ for (const file of files) {
340
+ // A generated mirror duplicates its stub on purpose.
341
+ if (file.generated)
342
+ continue;
343
+ for (const block of fileBlocks(file)) {
344
+ const key = normalise(block.text);
345
+ if (key.length < MIN_DUPLICATE_CHARS)
346
+ continue;
347
+ const list = seen.get(key) ?? [];
348
+ list.push({ file, block });
349
+ seen.set(key, list);
350
+ }
351
+ }
352
+ const out = [];
353
+ for (const occurrences of seen.values()) {
354
+ if (occurrences.length < 2)
355
+ continue;
356
+ const first = occurrences[0];
357
+ if (!first)
358
+ continue;
359
+ const wastedTokens = estimateTokens(first.block.text) * (occurrences.length - 1);
360
+ out.push({ occurrences, wastedTokens });
361
+ }
362
+ return out.sort((a, b) => b.wastedTokens - a.wastedTokens);
363
+ }
364
+ const mm005 = {
365
+ id: "MM005",
366
+ severity: "med",
367
+ title: "duplicated content across memory files",
368
+ run({ workspace }) {
369
+ const groups = findDuplicateBlocks(workspace.files);
370
+ const out = groups.map(({ occurrences, wastedTokens }) => {
371
+ const first = occurrences[0];
372
+ const where = occurrences
373
+ .slice(1)
374
+ .map((o) => `${o.file.rel}:${o.block.startLine}`)
375
+ .join(", ");
376
+ return {
377
+ rule: "MM005",
378
+ severity: "med",
379
+ message: `this block appears ${occurrences.length} times, also at ${where}`,
380
+ file: first.file.rel,
381
+ startLine: first.block.startLine,
382
+ endLine: first.block.endLine,
383
+ tokens: wastedTokens,
384
+ hint: "keep one copy and reference it; duplicates drift apart over time",
385
+ };
386
+ });
387
+ return out;
388
+ },
389
+ };
390
+ // ---------------------------------------------------------------------------
391
+ // MM006 OnDemandMemory list line exceeds the character cap
392
+ // ---------------------------------------------------------------------------
393
+ /**
394
+ * The hook is the descriptive text, with the list marker, any leading markdown link, and a
395
+ * separator dash stripped. That is the part a human writes and the part worth budgeting;
396
+ * the link is structural overhead the author cannot shorten much.
397
+ */
398
+ export function hookText(line) {
399
+ return line
400
+ .replace(/^\s*[-*+]\s+/, "")
401
+ .replace(/^\[[^\]]*\]\([^)]*\)\s*/, "")
402
+ .replace(/^[—–:-]\s*/, "")
403
+ .trim();
404
+ }
405
+ const mm006 = {
406
+ id: "MM006",
407
+ severity: "low",
408
+ title: "OnDemandMemory list line exceeds the character cap",
409
+ run({ workspace, config }) {
410
+ const lists = [workspace.onDemandList, workspace.autoMemoryList].filter((f, i, all) => f !== undefined && all.indexOf(f) === i);
411
+ const out = [];
412
+ for (const list of lists) {
413
+ list.lines.forEach((line, i) => {
414
+ if (!/^\s*[-*+]\s+/.test(line))
415
+ return;
416
+ const hook = hookText(line);
417
+ if (hook.length <= config.maxListLine)
418
+ return;
419
+ out.push({
420
+ rule: "MM006",
421
+ severity: "low",
422
+ message: `OnDemandMemory list hook is ${hook.length} characters, over the ${config.maxListLine} cap (line is ${line.length})`,
423
+ file: list.rel,
424
+ startLine: i + 1,
425
+ endLine: i + 1,
426
+ hint: "an OnDemandMemory list line is a hook, not a summary, and it is charged every turn",
427
+ });
428
+ });
429
+ }
430
+ return out;
431
+ },
432
+ };
433
+ // ---------------------------------------------------------------------------
434
+ // MM007 flat memory directory with no OnDemandMemory list
435
+ // ---------------------------------------------------------------------------
436
+ const mm007 = {
437
+ id: "MM007",
438
+ severity: "med",
439
+ title: "flat memory directory with no OnDemandMemory list",
440
+ run({ workspace }) {
441
+ if ((workspace.shape !== "memory-dir" && workspace.shape !== "auto-memory") || workspace.onDemandList)
442
+ return [];
443
+ if (workspace.files.length < 5)
444
+ return [];
445
+ const total = workspace.files.reduce((n, f) => n + f.tokens, 0);
446
+ return [
447
+ {
448
+ rule: "MM007",
449
+ severity: "med",
450
+ message: `${workspace.files.length} memory files with no OnDemandMemory list, so all ${formatTokens(total)} tokens get read`,
451
+ file: ".",
452
+ tokens: total,
453
+ hint: "add an OnDemandMemory list with one routing line per file so the agent can load on demand",
454
+ },
455
+ ];
456
+ },
457
+ };
458
+ // ---------------------------------------------------------------------------
459
+ // MM008 credential-shaped string in a memory file
460
+ // ---------------------------------------------------------------------------
461
+ const SECRETS = [
462
+ { re: /\bsk-ant-[A-Za-z0-9_-]{16,}/g, what: "an Anthropic API key" },
463
+ { re: /\bsk-[A-Za-z0-9]{32,}/g, what: "an OpenAI-style API key" },
464
+ { re: /\bnvapi-[A-Za-z0-9_-]{16,}/g, what: "an NVIDIA API key" },
465
+ { re: /\b(?:ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9]{20,}/g, what: "a GitHub token" },
466
+ { re: /\bgithub_pat_[A-Za-z0-9_]{20,}/g, what: "a GitHub fine-grained token" },
467
+ { re: /\bAKIA[0-9A-Z]{16}\b/g, what: "an AWS access key id" },
468
+ { re: /\bAIza[A-Za-z0-9_-]{35}\b/g, what: "a Google API key" },
469
+ { re: /\bxox[abprs]-[A-Za-z0-9-]{10,}/g, what: "a Slack token" },
470
+ // Stripe: underscore-separated, so the OpenAI-style `sk-` pattern above never matched these.
471
+ { re: /\b(?:sk|rk)_(?:live|test)_[A-Za-z0-9]{16,}/g, what: "a Stripe secret key" },
472
+ { re: /\bwhsec_[A-Za-z0-9]{16,}/g, what: "a Stripe webhook secret" },
473
+ { re: /-----BEGIN (?:RSA |EC |OPENSSH |PGP )?PRIVATE KEY-----/g, what: "a private key block" },
474
+ { re: /\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\b/g, what: "a JSON Web Token" },
475
+ { re: /\brnd_[A-Za-z0-9]{20,}\b/g, what: "a Render API key" },
476
+ ];
477
+ /** Never print a credential back out. Show just enough to locate it. */
478
+ function redact(match) {
479
+ const head = match.slice(0, Math.min(8, match.length));
480
+ return `${head}${"*".repeat(6)}`;
481
+ }
482
+ /** MM008 for one file. Exported so `init` can refuse to copy a credential into new files. */
483
+ export function secretFindings(file) {
484
+ const out = [];
485
+ file.lines.forEach((line, i) => {
486
+ for (const { re, what } of SECRETS) {
487
+ re.lastIndex = 0;
488
+ // Every match on the line, not just the first: two keys assigned on one line are two
489
+ // credentials to rotate, and reporting one of them reads as though the other is fine.
490
+ let m;
491
+ while ((m = re.exec(line)) !== null) {
492
+ if (!m[0])
493
+ break;
494
+ out.push({
495
+ rule: "MM008",
496
+ severity: "high",
497
+ message: `looks like ${what}: ${redact(m[0])}`,
498
+ file: file.rel,
499
+ startLine: i + 1,
500
+ endLine: i + 1,
501
+ hint: "memory files get committed and pasted into prompts; rotate it and remove it",
502
+ });
503
+ }
504
+ }
505
+ });
506
+ return out;
507
+ }
508
+ const mm008 = {
509
+ id: "MM008",
510
+ severity: "high",
511
+ title: "credential-shaped string in a memory file",
512
+ run({ workspace }) {
513
+ return workspace.files.flatMap(secretFindings);
514
+ },
515
+ };
516
+ // ---------------------------------------------------------------------------
517
+ // ---------------------------------------------------------------------------
518
+ // MM009 a flat section over the granularity threshold in a long-term file
519
+ // ---------------------------------------------------------------------------
520
+ export { GRANULARITY_TOKENS };
521
+ /**
522
+ * sections() yields leaves: a section ends where the next heading of any level starts, so a
523
+ * leaf this large has no subheading by construction. An episodic section (mostly dated
524
+ * bullets) is exempt because the router already addresses it one entry at a time (O4).
525
+ * The 2026-09-04 audit found a 4,467-token "Current state" with no subheading; any recall
526
+ * that touched it paid all of it.
527
+ */
528
+ const mm009 = {
529
+ id: "MM009",
530
+ severity: "low",
531
+ title: "flat section over the granularity threshold",
532
+ run({ workspace }) {
533
+ const out = [];
534
+ for (const file of workspace.files) {
535
+ if (file.alwaysLoaded || file === workspace.onDemandList || file === workspace.autoMemoryList)
536
+ continue;
537
+ if (file.generated || file.kind === "alwaysOn" || file.kind === "onDemandList")
538
+ continue;
539
+ for (const s of sections(file.lines)) {
540
+ if (s.level === 0)
541
+ continue;
542
+ const tokens = estimateTokens(file.lines.slice(s.startLine - 1, s.endLine).join("\n"));
543
+ if (tokens < GRANULARITY_TOKENS)
544
+ continue;
545
+ const body = file.lines.slice(s.startLine, s.endLine);
546
+ if (isEpisodicBody(body))
547
+ continue;
548
+ out.push({
549
+ rule: "MM009",
550
+ severity: "low",
551
+ message: `"${s.heading}" is ${formatTokens(tokens)} tokens with no subheading, so a recall that touches it pays all of it`,
552
+ file: file.rel,
553
+ startLine: s.startLine,
554
+ endLine: s.endLine,
555
+ tokens,
556
+ hint: "add H3 subheadings so retrieval can return the part the task needs (O7)",
557
+ });
558
+ }
559
+ }
560
+ return out.sort((a, b) => (b.tokens ?? 0) - (a.tokens ?? 0));
561
+ },
562
+ };
563
+ // ---------------------------------------------------------------------------
564
+ // MM010 a compiled workspace has drifted from what init wrote
565
+ // ---------------------------------------------------------------------------
566
+ /** CRLF-normalised, because a checkout can rewrite line endings without anyone editing. */
567
+ function normalised(content) {
568
+ return content.replace(/\r\n/g, "\n");
569
+ }
570
+ /**
571
+ * The re-apply case: someone comes back to a memory they optimised earlier. `doctor` is the
572
+ * front door for that, so it compares every OnDemandMemory file, the AlwaysOnMemory body and the
573
+ * OnDemandMemory list on disk against the hashes `init` recorded in manifest.json and names each
574
+ * file that moved, was removed, or appeared. The remedy is always the same and is printed with
575
+ * every finding. Read-only, as every rule is; the writing hand stays `init --update`.
576
+ *
577
+ * OnDemandMemory file hashes were taken over the trimmed content and the file is written with
578
+ * one trailing newline, so the OnDemandMemory file side compares trimmed; AlwaysOnMemory.md was
579
+ * written verbatim.
580
+ */
581
+ const mm010 = {
582
+ id: "MM010",
583
+ severity: "med",
584
+ title: "compiled workspace has drifted since init",
585
+ run({ workspace }) {
586
+ const manifest = workspace.manifest;
587
+ if (workspace.shape !== "compiled" || !manifest)
588
+ return [];
589
+ const out = [];
590
+ const remedy = "re-apply with: minnimemory init --update (keeps every OnDemandMemory file, rewrites the OnDemandMemory list, manifest and stub)";
591
+ const byRel = new Map(workspace.files.map((f) => [f.rel.replace(/\\/g, "/"), f]));
592
+ const compiledDir = COMPILED_DIR_NAME;
593
+ const seen = new Set();
594
+ for (const m of manifest.onDemandFiles) {
595
+ const rel = `${compiledDir}/${m.file}`;
596
+ seen.add(rel);
597
+ const file = byRel.get(rel);
598
+ if (!file) {
599
+ out.push({ rule: "MM010", severity: "med", message: `${rel} is in the manifest but missing on disk`, file: rel, hint: remedy });
600
+ continue;
601
+ }
602
+ if (driftHash(file.content) !== m.hash) {
603
+ out.push({
604
+ rule: "MM010",
605
+ severity: "med",
606
+ message: `${rel} was edited since init wrote it`,
607
+ file: rel,
608
+ tokens: file.tokens,
609
+ hint: remedy,
610
+ });
611
+ }
612
+ }
613
+ for (const f of workspace.files) {
614
+ const rel = f.rel.replace(/\\/g, "/");
615
+ if (f.kind === "onDemand" && !seen.has(rel)) {
616
+ out.push({ rule: "MM010", severity: "med", message: `${rel} is on disk but not in the manifest`, file: rel, tokens: f.tokens, hint: remedy });
617
+ }
618
+ }
619
+ {
620
+ const entry = manifest.always;
621
+ const rel = `${compiledDir}/${entry.file}`;
622
+ const file = byRel.get(rel);
623
+ if (file && verbatimHash(normalised(file.content)) !== entry.hash) {
624
+ out.push({ rule: "MM010", severity: "med", message: `${rel} was edited since init wrote it`, file: rel, tokens: file.tokens, hint: remedy });
625
+ }
626
+ }
627
+ // The stub is the always-loaded prefix, so content appended to it is the drift that costs
628
+ // most: it is paid on every turn and routed by nothing. Without this check doctor reported
629
+ // "no drift since init. Nothing to re-apply." with un-routed content sitting in the host
630
+ // file, and MM001 only noticed once the regrown prefix crossed the budget.
631
+ if (manifest.stub) {
632
+ const rel = manifest.stub.file;
633
+ const file = byRel.get(rel);
634
+ if (file && verbatimHash(normalised(file.content)) !== manifest.stub.hash) {
635
+ out.push({
636
+ rule: "MM010",
637
+ severity: "med",
638
+ message: `${rel} was edited since init wrote it`,
639
+ file: rel,
640
+ tokens: file.tokens,
641
+ hint: remedy,
642
+ });
643
+ }
644
+ }
645
+ return out;
646
+ },
647
+ };
648
+ export const RULES = [mm001, mm002, mm003, mm004, mm005, mm006, mm007, mm008, mm009, mm010];
649
+ export function ruleById(id) {
650
+ return RULES.find((r) => r.id.toLowerCase() === id.toLowerCase());
651
+ }