minnimemory 1.0.0-beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/LICENSE +39 -0
  2. package/README.md +824 -0
  3. package/dist/bench.d.ts +98 -0
  4. package/dist/bench.js +142 -0
  5. package/dist/benchReport.d.ts +12 -0
  6. package/dist/benchReport.js +128 -0
  7. package/dist/bounds.d.ts +40 -0
  8. package/dist/bounds.js +44 -0
  9. package/dist/cli.d.ts +15 -0
  10. package/dist/cli.js +503 -0
  11. package/dist/compile.d.ts +187 -0
  12. package/dist/compile.js +516 -0
  13. package/dist/discover.d.ts +125 -0
  14. package/dist/discover.js +520 -0
  15. package/dist/doctor.d.ts +9 -0
  16. package/dist/doctor.js +67 -0
  17. package/dist/episodic.d.ts +47 -0
  18. package/dist/episodic.js +130 -0
  19. package/dist/hook.d.ts +45 -0
  20. package/dist/hook.js +104 -0
  21. package/dist/index.d.ts +18 -0
  22. package/dist/index.js +18 -0
  23. package/dist/init.d.ts +125 -0
  24. package/dist/init.js +475 -0
  25. package/dist/instructions.d.ts +60 -0
  26. package/dist/instructions.js +270 -0
  27. package/dist/mcp.d.ts +109 -0
  28. package/dist/mcp.js +252 -0
  29. package/dist/mcpServer.d.ts +136 -0
  30. package/dist/mcpServer.js +997 -0
  31. package/dist/paths.d.ts +25 -0
  32. package/dist/paths.js +47 -0
  33. package/dist/recall.d.ts +113 -0
  34. package/dist/recall.js +256 -0
  35. package/dist/recallDir.d.ts +50 -0
  36. package/dist/recallDir.js +187 -0
  37. package/dist/reorganize.d.ts +62 -0
  38. package/dist/reorganize.js +216 -0
  39. package/dist/report.d.ts +16 -0
  40. package/dist/report.js +204 -0
  41. package/dist/router.d.ts +141 -0
  42. package/dist/router.js +314 -0
  43. package/dist/rules.d.ts +32 -0
  44. package/dist/rules.js +651 -0
  45. package/dist/scan.d.ts +110 -0
  46. package/dist/scan.js +173 -0
  47. package/dist/text.d.ts +158 -0
  48. package/dist/text.js +395 -0
  49. package/dist/tokenizer.d.ts +26 -0
  50. package/dist/tokenizer.js +69 -0
  51. package/dist/types.d.ts +156 -0
  52. package/dist/types.js +17 -0
  53. package/dist/version.d.ts +7 -0
  54. package/dist/version.js +7 -0
  55. package/dist/writeProtocol.d.ts +19 -0
  56. package/dist/writeProtocol.js +45 -0
  57. package/examples/CLAUDE.md +75 -0
  58. package/examples/README.md +7 -0
  59. package/package.json +52 -0
@@ -0,0 +1,516 @@
1
+ /**
2
+ * The compiler: turn one memory file into a cache-stable AlwaysOnMemory body plus routed
3
+ * OnDemandMemory files.
4
+ *
5
+ * Pure by design. Everything here is string in, strings out, so the whole compile can be
6
+ * tested from fixtures and diffed before anything touches a user's disk.
7
+ *
8
+ * Structure follows the tiered-memory pattern already running in production in the Minni
9
+ * agents (AlwaysOnMemory.md, OnDemandMemory/, fixed assembly order), and the emitted
10
+ * instruction block comes from the MinniMemory v1.0 research set. See instructions.ts.
11
+ */
12
+ import { createHash } from "node:crypto";
13
+ import { defaultProfile, profile as profileFor, renderInstructions } from "./instructions.js";
14
+ import { classifyKind, episodicFromMarkdown, episodicToJson } from "./episodic.js";
15
+ import { renderWriteProtocolFile, WRITE_PROTOCOL_HEADING, WRITE_PROTOCOL_FILE_NAME, WRITE_PROTOCOL_TRIGGERS, } from "./writeProtocol.js";
16
+ import { blocks, GRANULARITY_TOKENS, LIST_END, LIST_START, INSTRUCTIONS_END, INSTRUCTIONS_START, keywords, sections, sectionVolatilityReasons, slugify, splitFrontmatter, volatilityReasons, } from "./text.js";
17
+ import { selectTriggers } from "./router.js";
18
+ import { estimateTokens } from "./tokenizer.js";
19
+ /** Marker identifying a host file this tool generated, so it is never double counted. */
20
+ export const STUB_MARKER = "<!-- minnimemory:stub v1 -->";
21
+ /** Default always-loaded budget in tokens; matches doctor's MM001 default. */
22
+ export const DEFAULT_BUDGET = 2000;
23
+ /**
24
+ * Rough cost of one OnDemandMemory list line and of the fixed list scaffolding (markers,
25
+ * heading, and the routing line the `none` profile adds), used only to keep AlwaysOnMemory
26
+ * placement inside the budget. Measured with approx-v2 on the list-form OnDemandMemory list
27
+ * (2026-09-04 audit); the previous table form cost 159 fixed while the estimator assumed 90.
28
+ */
29
+ const LIST_LINE_COST = 22;
30
+ const LIST_BASE_COST = 40;
31
+ /** Headings whose content is identity or invariant, so it belongs in AlwaysOnMemory. */
32
+ export const ALWAYS_ON_HINTS = /\b(identity|overview|about|purpose|mission|principles?|conventions?|rules?|constraints?|boundar\w*|guardrails?|guidelines?|style|tone|standards?|polic(?:y|ies)|invariants?|philosophy|decisions?)\b/i;
33
+ /**
34
+ * Below this, an OnDemandMemory file is not worth routing to: the list line and the decision
35
+ * to load it cost more than just having read the content. Consecutive small sections are merged.
36
+ */
37
+ const MIN_ON_DEMAND_TOKENS = 400;
38
+ function hash(text) {
39
+ return createHash("sha256").update(text).digest("hex").slice(0, 16);
40
+ }
41
+ /**
42
+ * The hash MM010 and recall's drift check both compare against a manifest entry.
43
+ *
44
+ * CRLF-normalised and trailing-whitespace-trimmed on purpose: a checkout can rewrite line
45
+ * endings with nobody editing anything, and a file that differs only that way has not drifted.
46
+ * One copy, because the dated-bullet regex taught us what three copies of a rule costs
47
+ * (2026-09-13 audit, D7).
48
+ */
49
+ export function driftHash(content) {
50
+ return hash(content.replace(/\r\n/g, "\n").trimEnd());
51
+ }
52
+ /**
53
+ * Shared with rules.ts's MM010 check against AlwaysOnMemory.md and the stub: sha256 slice 16
54
+ * with no normalisation, the caller's job (CRLF-only, no trimEnd - those files are written
55
+ * verbatim, so trimming would blur an edit driftHash would forgive on purpose for OnDemandMemory
56
+ * files but should not here).
57
+ */
58
+ export { hash as verbatimHash };
59
+ /** Pick the heading level that actually divides the document into topics. */
60
+ function detectSplitLevel(all) {
61
+ for (let level = 2; level <= 4; level++) {
62
+ if (all.filter((s) => s.level === level).length >= 2)
63
+ return level;
64
+ }
65
+ // A document of only H1s still needs splitting somewhere.
66
+ if (all.filter((s) => s.level === 1).length >= 2)
67
+ return 1;
68
+ return 2;
69
+ }
70
+ function textOf(lines, startLine, endLine) {
71
+ return lines.slice(startLine - 1, endLine).join("\n").trimEnd();
72
+ }
73
+ /**
74
+ * Choose each OnDemandMemory file's list triggers against the whole file set (router.ts
75
+ * selectTriggers): frequency here times rarity elsewhere, headings weighted. Replaces the
76
+ * frequency-only keywords() pick plus a post-hoc prune, which produced `bars, free, merged` for
77
+ * a trading-app memory file (audit 2026-09-04). The per-file keywords() call remains the
78
+ * fallback for paths that see one file at a time (init --update).
79
+ */
80
+ function assignTriggers(onDemandFiles) {
81
+ const chosen = selectTriggers(onDemandFiles.map((m) => ({ name: m.name, heading: m.heading, content: m.content })));
82
+ for (const m of onDemandFiles)
83
+ m.triggers = chosen.get(m.name) ?? m.triggers;
84
+ }
85
+ /**
86
+ * Merge runs of small consecutive sections into OnDemandMemory files worth routing to.
87
+ * Order is preserved, so the fixed assembly order still matches document order.
88
+ */
89
+ function mergeSmall(onDemandFiles, min) {
90
+ const out = [];
91
+ let bucket = [];
92
+ const flush = () => {
93
+ if (bucket.length === 0)
94
+ return;
95
+ const first = bucket[0];
96
+ if (!first) {
97
+ bucket = [];
98
+ return;
99
+ }
100
+ if (bucket.length === 1) {
101
+ out.push(first);
102
+ bucket = [];
103
+ return;
104
+ }
105
+ const last = bucket[bucket.length - 1] ?? first;
106
+ const content = bucket.map((b) => b.content).join("\n\n");
107
+ const triggers = [...new Set(bucket.flatMap((b) => b.triggers))].slice(0, 8);
108
+ // A merged OnDemandMemory file must not inherit the first section's name: a file called
109
+ // current_status.md that also holds testing and deployment misleads the reader
110
+ // and the agent about what is inside it.
111
+ const name = slugify(`${first.heading} to ${last.heading}`);
112
+ out.push({
113
+ name,
114
+ file: `${ON_DEMAND_DIR_NAME}/${name}.md`,
115
+ heading: `${first.heading} through ${last.heading}`,
116
+ content,
117
+ tokens: estimateTokens(content),
118
+ triggers,
119
+ sourceStartLine: first.sourceStartLine,
120
+ sourceEndLine: last.sourceEndLine,
121
+ reason: bucket.some((b) => b.reason === "volatile")
122
+ ? "volatile"
123
+ : bucket.some((b) => b.reason === "over-budget")
124
+ ? "over-budget"
125
+ : "task-specific",
126
+ kind: first.kind,
127
+ });
128
+ bucket = [];
129
+ };
130
+ for (const m of onDemandFiles) {
131
+ // A file the granularity split produced stands alone: flush whatever was accumulating,
132
+ // emit it untouched, and start a fresh bucket after it. So does an episodic section (O2):
133
+ // it is append-only and may become JSON (O3), so it never shares a file with prose.
134
+ if (m.split || m.kind === "episodic") {
135
+ flush();
136
+ out.push(m);
137
+ continue;
138
+ }
139
+ // Kinds are edited differently (O2), so a bucket holds one kind only.
140
+ if (bucket.length > 0 && bucket[0]?.kind !== m.kind)
141
+ flush();
142
+ bucket.push(m);
143
+ if (bucket.reduce((n, b) => n + b.tokens, 0) >= min)
144
+ flush();
145
+ }
146
+ flush();
147
+ return out;
148
+ }
149
+ /**
150
+ * The list title is deliberately not derived from the document title: a title can carry a
151
+ * date or a status word ("Minni (renamed 2026-08-11)"), and the list must stay free of anything
152
+ * the volatility rules would flag. The whole list is fenced with markers so `doctor` can tell a
153
+ * routing manifest from prose.
154
+ *
155
+ * One line per OnDemandMemory file and nothing that restates a Part 2 rule (P4). Rules 15, 16,
156
+ * 22 and 23 already say how to use the list and sit in the same prefix whenever a discipline
157
+ * block is embedded; the previous table-plus-prose form repeated them at 159 tokens per compile.
158
+ * Only the `none` profile, which embeds no block, gets one routing line after the list. Assembly
159
+ * order is the manifest's `order`, not prefix text.
160
+ */
161
+ /** The directory `init` writes next to the host file. Named here, where the stub's
162
+ * OnDemandMemory list is rendered, so the two can never disagree about where a file lives. */
163
+ export const COMPILED_DIR_NAME = ".minnimemory";
164
+ /** The one always-loaded file inside the compiled directory: AlwaysOnMemory body plus the
165
+ * OnDemandMemory list. */
166
+ export const ALWAYS_ON_FILE_NAME = "AlwaysOnMemory.md";
167
+ /** The OnDemandMemory files directory inside the compiled directory. */
168
+ export const ON_DEMAND_DIR_NAME = "OnDemandMemory";
169
+ function renderOnDemandList(onDemandFiles, withRoutingLine, pathPrefix = "") {
170
+ const lines = [LIST_START, "# OnDemandMemory"];
171
+ if (onDemandFiles.length === 0) {
172
+ lines.push("No OnDemandMemory files. Everything lives in AlwaysOnMemory.", LIST_END, "");
173
+ return lines.join("\n");
174
+ }
175
+ // Paths are relative to the file the list sits in. AlwaysOnMemory.md lives inside the
176
+ // compiled directory beside OnDemandMemory/; the stub lives one level up, in the project, and
177
+ // an agent that reads `OnDemandMemory/x.md` from there finds nothing. The 2026-09-05 session
178
+ // measurement caught exactly that: the agent tried the bare path, missed, and gave up.
179
+ for (const m of onDemandFiles) {
180
+ lines.push(`- \`${pathPrefix}${m.file}\`: ${m.triggers.join(", ")}`);
181
+ }
182
+ if (withRoutingLine) {
183
+ lines.push("", "Read an OnDemandMemory file only when the task needs it, and answer from it, never from this list.");
184
+ }
185
+ lines.push(LIST_END, "");
186
+ return lines.join("\n");
187
+ }
188
+ /** The instruction block, fenced so `doctor` skips it and `init --update` can strip it. */
189
+ function renderInstructionRegion(profile) {
190
+ const body = renderInstructions(profile);
191
+ if (!body)
192
+ return "";
193
+ return [INSTRUCTIONS_START, body.trimEnd(), INSTRUCTIONS_END].join("\n");
194
+ }
195
+ function renderAlwaysOnBody(frontmatter, title, preamble, alwaysOnSections, profile) {
196
+ const parts = [];
197
+ if (frontmatter)
198
+ parts.push(frontmatter, "");
199
+ parts.push(`# ${title}`, "");
200
+ if (preamble.trim())
201
+ parts.push(preamble.trim(), "");
202
+ for (const s of alwaysOnSections)
203
+ parts.push(s.trim(), "");
204
+ const instructions = renderInstructionRegion(profile);
205
+ if (instructions)
206
+ parts.push(instructions, "");
207
+ return parts.join("\n");
208
+ }
209
+ /**
210
+ * Frontmatter, when present, stays the very first thing in the stub: a YAML header that is not
211
+ * on line 1 stops being a header. The generated-file marker follows it. `alwaysOnBody` here is
212
+ * the AlwaysOnMemory body without its frontmatter, since the stub places the frontmatter itself.
213
+ */
214
+ function renderStub(frontmatter, alwaysOnBody, onDemandList, onDemandCount) {
215
+ const parts = [];
216
+ if (frontmatter)
217
+ parts.push(frontmatter);
218
+ parts.push(STUB_MARKER, "<!-- generated by minnimemory; edit .minnimemory/ instead -->", "");
219
+ parts.push(alwaysOnBody.trimEnd(), "");
220
+ if (onDemandCount > 0)
221
+ parts.push(onDemandList.trimEnd(), "");
222
+ return parts.join("\n");
223
+ }
224
+ /**
225
+ * `.minnimemory/AlwaysOnMemory.md`: the AlwaysOnMemory body (already carries its own
226
+ * frontmatter) plus the OnDemandMemory list, the same join the stub uses, written to its own
227
+ * file instead of into the host file.
228
+ */
229
+ function renderAlwaysOn(alwaysOnBody, onDemandList, onDemandCount) {
230
+ const parts = [alwaysOnBody.trimEnd(), ""];
231
+ if (onDemandCount > 0)
232
+ parts.push(onDemandList.trimEnd(), "");
233
+ return parts.join("\n");
234
+ }
235
+ /**
236
+ * Is this host file a stub this tool generated? Only a marker on the first line after any
237
+ * frontmatter counts. A marker quoted in a code block, or anywhere else in prose, is text
238
+ * (security audit 2026-09-02, finding 5).
239
+ */
240
+ export function isStub(source) {
241
+ const lines = source.split(/\r?\n/);
242
+ const { bodyStartLine } = splitFrontmatter(lines);
243
+ return (lines[bodyStartLine - 1] ?? "").trim() === STUB_MARKER;
244
+ }
245
+ export function compile(source, options) {
246
+ const lines = source.split(/\r?\n/);
247
+ const { frontmatter: frontmatterLines, bodyStartLine } = splitFrontmatter(lines);
248
+ const frontmatter = frontmatterLines.join("\n");
249
+ const all = sections(lines).filter((s) => s.startLine >= bodyStartLine);
250
+ const profile = options.profile ?? (options.profileName ? profileFor(options.profileName) : defaultProfile());
251
+ const splitLevel = options.splitLevel ?? detectSplitLevel(all);
252
+ const splits = all.filter((s) => s.level === splitLevel);
253
+ const firstSplitLine = splits[0]?.startLine ?? lines.length + 1;
254
+ // The title is an H1 that comes before the first split heading. An H1 further down is a
255
+ // section like any other, not the document's name.
256
+ const titleSection = all.find((s) => s.level === 1 && s.startLine < firstSplitLine);
257
+ const title = titleSection?.heading ?? options.sourceName.replace(/\.md$/i, "");
258
+ // Everything above the first split heading is preamble. That includes any prose that sits
259
+ // above the title line, so nothing written before the H1 is lost.
260
+ const preambleStart = bodyStartLine;
261
+ const preambleLines = titleSection
262
+ ? [
263
+ ...lines.slice(bodyStartLine - 1, titleSection.startLine - 1),
264
+ ...lines.slice(titleSection.startLine, firstSplitLine - 1),
265
+ ]
266
+ : lines.slice(bodyStartLine - 1, firstSplitLine - 1);
267
+ const preamble = preambleLines.join("\n").trimEnd();
268
+ const alwaysOnSections = [];
269
+ const onDemandFiles = [];
270
+ // The preamble is not exempt from the cache-stability law. Split it per block and route the
271
+ // volatile blocks out, so a dated audit note cannot sit in the always-loaded prefix just
272
+ // because it happened to appear above the first heading.
273
+ const preambleBlocks = blocks(preamble.split("\n"));
274
+ const stablePreamble = preambleBlocks.filter((b) => volatilityReasons(b.text).length === 0);
275
+ const volatilePreamble = preambleBlocks.filter((b) => volatilityReasons(b.text).length > 0);
276
+ if (volatilePreamble.length > 0) {
277
+ const content = ["## Notes", "", ...volatilePreamble.map((b) => b.text)].join("\n\n");
278
+ onDemandFiles.push({
279
+ name: "notes",
280
+ file: `${ON_DEMAND_DIR_NAME}/notes.md`,
281
+ heading: "Notes",
282
+ content,
283
+ tokens: estimateTokens(content),
284
+ triggers: keywords("notes status", content),
285
+ sourceStartLine: preambleStart,
286
+ sourceEndLine: firstSplitLine - 1,
287
+ reason: "volatile",
288
+ });
289
+ }
290
+ const preambleText = stablePreamble.map((b) => b.text).join("\n\n");
291
+ const entries = splits.map((s, i) => {
292
+ const next = splits[i + 1];
293
+ const endLine = next ? next.startLine - 1 : lines.length;
294
+ const content = textOf(lines, s.startLine, endLine);
295
+ const volatile = sectionVolatilityReasons(content.split("\n"));
296
+ // The cache-stability law: volatile content never enters the always-loaded prefix,
297
+ // no matter how always-on-like its heading reads.
298
+ const alwaysOnCandidate = ALWAYS_ON_HINTS.test(s.heading) && volatile.length === 0;
299
+ return { section: s, content, endLine, tokens: estimateTokens(content), volatile, alwaysOnCandidate };
300
+ });
301
+ // The stable preamble competes for AlwaysOnMemory like any other candidate. Real AGENTS.md
302
+ // files put a thousand tokens of prose before the first H2; exempting that from the budget
303
+ // would push every actual rule section out instead.
304
+ if (preambleText.trim()) {
305
+ entries.unshift({
306
+ section: { heading: title, level: 0, startLine: preambleStart, endLine: firstSplitLine - 1 },
307
+ content: preambleText,
308
+ endLine: firstSplitLine - 1,
309
+ tokens: estimateTokens(preambleText),
310
+ volatile: [],
311
+ alwaysOnCandidate: true,
312
+ preamble: true,
313
+ });
314
+ }
315
+ // Budget-aware placement. A heading can read as always-on ("Architecture guidelines", "Code
316
+ // Style") and still be thousands of tokens; keeping every such section in AlwaysOnMemory just
317
+ // moves the MM001 finding from the source to the compiled output. Smallest candidates are kept
318
+ // first, because a short set of invariants is what AlwaysOnMemory is for, and whatever does
319
+ // not fit is routed to an OnDemandMemory file with the reason recorded so the user can see it
320
+ // and raise --budget.
321
+ const budget = options.budget ?? DEFAULT_BUDGET;
322
+ const candidates = entries.filter((e) => e.alwaysOnCandidate).sort((a, b) => a.tokens - b.tokens);
323
+ const onDemandEstimate = entries.length - candidates.length + (volatilePreamble.length > 0 ? 1 : 0);
324
+ let used = estimateTokens(renderAlwaysOnBody("", title, "", [], profile)) + LIST_BASE_COST + LIST_LINE_COST * onDemandEstimate;
325
+ const kept = new Set();
326
+ for (const c of candidates) {
327
+ if (used + c.tokens <= budget) {
328
+ kept.add(c);
329
+ used += c.tokens;
330
+ }
331
+ }
332
+ const demoted = candidates.filter((c) => !kept.has(c));
333
+ const stableText = entries.find((e) => e.preamble && kept.has(e))?.content ?? "";
334
+ for (const e of entries) {
335
+ if (kept.has(e)) {
336
+ if (!e.preamble)
337
+ alwaysOnSections.push(e.content);
338
+ continue;
339
+ }
340
+ if (e.preamble) {
341
+ const name = slugify(`${title} overview`);
342
+ onDemandFiles.push({
343
+ name,
344
+ file: `${ON_DEMAND_DIR_NAME}/${name}.md`,
345
+ heading: `${title} overview`,
346
+ content: `## ${title} overview\n\n${e.content}`,
347
+ tokens: e.tokens,
348
+ triggers: keywords(`${title} overview`, e.content),
349
+ sourceStartLine: e.section.startLine,
350
+ sourceEndLine: e.endLine,
351
+ reason: "over-budget",
352
+ });
353
+ continue;
354
+ }
355
+ const reason = e.volatile.length > 0 ? "volatile" : e.alwaysOnCandidate ? "over-budget" : "task-specific";
356
+ // The granularity clause (O7) on the agent-driven route: an agent that reads a whole file
357
+ // the OnDemandMemory list named pays the whole file, so a section over the threshold is
358
+ // emitted as one OnDemandMemory file per subheading, the parent heading and its intro lines
359
+ // first. Every line lands in exactly one file, in document order, so losslessness and the
360
+ // assembly order both hold. A flat section has nothing to split on; doctor's MM009 says so.
361
+ const subs = e.tokens >= GRANULARITY_TOKENS
362
+ ? all.filter((s) => s.level === splitLevel + 1 && s.startLine > e.section.startLine && s.startLine <= e.endLine)
363
+ : [];
364
+ if (subs.length > 0) {
365
+ const first = subs[0];
366
+ const introEnd = first.startLine - 1;
367
+ const intro = textOf(lines, e.section.startLine, introEnd);
368
+ const introTokens = estimateTokens(intro);
369
+ // A parent heading plus a sentence or two is not worth its own list line, so it rides
370
+ // with the first subheading instead. Only a substantial intro becomes its own file.
371
+ const introStandsAlone = introTokens >= LIST_LINE_COST * 2;
372
+ if (introStandsAlone) {
373
+ onDemandFiles.push({
374
+ name: slugify(e.section.heading),
375
+ file: `${ON_DEMAND_DIR_NAME}/${slugify(e.section.heading)}.md`,
376
+ heading: e.section.heading,
377
+ content: intro,
378
+ tokens: introTokens,
379
+ triggers: keywords(e.section.heading, intro),
380
+ sourceStartLine: e.section.startLine,
381
+ sourceEndLine: introEnd,
382
+ reason,
383
+ split: true,
384
+ });
385
+ }
386
+ subs.forEach((sub, i) => {
387
+ const next = subs[i + 1];
388
+ const end = next ? next.startLine - 1 : e.endLine;
389
+ const isFirst = i === 0;
390
+ const startLine = !introStandsAlone && isFirst ? e.section.startLine : sub.startLine;
391
+ const content = textOf(lines, startLine, end);
392
+ const heading = `${e.section.heading}: ${sub.heading}`;
393
+ const name = slugify(heading);
394
+ onDemandFiles.push({
395
+ name,
396
+ file: `${ON_DEMAND_DIR_NAME}/${name}.md`,
397
+ heading,
398
+ content,
399
+ tokens: estimateTokens(content),
400
+ triggers: keywords(heading, content),
401
+ sourceStartLine: startLine,
402
+ sourceEndLine: end,
403
+ reason,
404
+ split: true,
405
+ });
406
+ });
407
+ continue;
408
+ }
409
+ const name = slugify(e.section.heading);
410
+ onDemandFiles.push({
411
+ name,
412
+ file: `${ON_DEMAND_DIR_NAME}/${name}.md`,
413
+ heading: e.section.heading,
414
+ content: e.content,
415
+ tokens: e.tokens,
416
+ triggers: keywords(e.section.heading, e.content),
417
+ sourceStartLine: e.section.startLine,
418
+ sourceEndLine: e.endLine,
419
+ reason,
420
+ });
421
+ }
422
+ // O2 before merging: the kind decides what may share a file.
423
+ for (const m of onDemandFiles)
424
+ m.kind = m.kind ?? classifyKind(m.heading, m.content.split("\n").slice(1));
425
+ const merged = options.onDemandFilesOverride ?? mergeSmall(onDemandFiles, MIN_ON_DEMAND_TOKENS);
426
+ // Slugs must be unique or OnDemandMemory files overwrite each other on disk.
427
+ const seen = new Map();
428
+ for (const m of merged) {
429
+ const n = seen.get(m.name) ?? 0;
430
+ seen.set(m.name, n + 1);
431
+ if (n > 0) {
432
+ m.name = `${m.name}_${n + 1}`;
433
+ m.file = `${ON_DEMAND_DIR_NAME}/${m.name}.md`;
434
+ }
435
+ }
436
+ assignTriggers(merged);
437
+ // O2: every OnDemandMemory file carries its kind. O3: an episodic file is written as JSON
438
+ // when asked, and its on-disk bytes are what gets hashed and counted, so tokens follow the JSON.
439
+ for (const m of merged) {
440
+ if (m.generated)
441
+ continue;
442
+ m.kind = m.kind ?? classifyKind(m.heading, m.content.split("\n").slice(1));
443
+ if (options.episodicJson && m.kind === "episodic" && m.file.endsWith(".md")) {
444
+ m.content = episodicToJson(episodicFromMarkdown(m.content)).trimEnd();
445
+ m.file = m.file.replace(/\.md$/, ".json");
446
+ m.tokens = estimateTokens(m.content);
447
+ }
448
+ }
449
+ // O9: the write protocol rides as a routed OnDemandMemory file, never in the prefix. Appended
450
+ // last so the source OnDemandMemory files keep document order, and skipped when an override
451
+ // already has it.
452
+ if (options.writeProtocol !== false && !merged.some((m) => m.name === WRITE_PROTOCOL_FILE_NAME)) {
453
+ const content = renderWriteProtocolFile();
454
+ merged.push({
455
+ name: WRITE_PROTOCOL_FILE_NAME,
456
+ file: `${ON_DEMAND_DIR_NAME}/${WRITE_PROTOCOL_FILE_NAME}.md`,
457
+ heading: WRITE_PROTOCOL_HEADING,
458
+ content,
459
+ tokens: estimateTokens(content),
460
+ triggers: [...WRITE_PROTOCOL_TRIGGERS],
461
+ sourceStartLine: 0,
462
+ sourceEndLine: 0,
463
+ reason: "generated",
464
+ kind: "procedural",
465
+ generated: true,
466
+ });
467
+ }
468
+ const stubBody = renderAlwaysOnBody("", title, stableText, alwaysOnSections, profile);
469
+ const alwaysOnBody = renderAlwaysOnBody(frontmatter, title, stableText, alwaysOnSections, profile);
470
+ const onDemandList = renderOnDemandList(merged, profile.length === 0);
471
+ const stubOnDemandList = renderOnDemandList(merged, profile.length === 0, `${COMPILED_DIR_NAME}/`);
472
+ const stub = renderStub(frontmatter, stubBody, stubOnDemandList, merged.length);
473
+ const alwaysOn = renderAlwaysOn(alwaysOnBody, onDemandList, merged.length);
474
+ return {
475
+ title,
476
+ sourceName: options.sourceName,
477
+ sourceHash: hash(source),
478
+ frontmatter,
479
+ alwaysOnBody,
480
+ onDemandList,
481
+ alwaysOn,
482
+ stub,
483
+ onDemandFiles: merged,
484
+ before: estimateTokens(source),
485
+ after: estimateTokens(stub),
486
+ instructionTokens: estimateTokens(renderInstructions(profile)),
487
+ budget,
488
+ demoted: demoted
489
+ .sort((a, b) => a.section.startLine - b.section.startLine)
490
+ .map((d) => ({ heading: d.section.heading, tokens: d.tokens })),
491
+ };
492
+ }
493
+ export function buildManifest(compiled, tokenizer) {
494
+ return {
495
+ version: 4,
496
+ tokenizer,
497
+ generatedFrom: compiled.sourceName,
498
+ sourceHash: compiled.sourceHash,
499
+ instructionStatus: "output-side rules measured 2026-09-05; context-side rules unmeasured; see the Measurement policy section of the MinniMemoryMCP README",
500
+ order: compiled.onDemandFiles.map((m) => m.name),
501
+ prefix: { before: compiled.before, after: compiled.after },
502
+ always: { file: ALWAYS_ON_FILE_NAME, tokens: estimateTokens(compiled.alwaysOn), hash: hash(compiled.alwaysOn) },
503
+ stub: { file: compiled.sourceName, tokens: estimateTokens(compiled.stub), hash: hash(compiled.stub) },
504
+ onDemandFiles: compiled.onDemandFiles.map((m) => ({
505
+ name: m.name,
506
+ file: m.file,
507
+ tokens: m.tokens,
508
+ hash: driftHash(m.content),
509
+ triggers: m.triggers,
510
+ sourceLines: [m.sourceStartLine, m.sourceEndLine],
511
+ reason: m.reason,
512
+ kind: m.kind,
513
+ generated: m.generated ?? false,
514
+ })),
515
+ };
516
+ }
@@ -0,0 +1,125 @@
1
+ /**
2
+ * Work out what kind of memory setup we are pointed at, and which files the model
3
+ * actually pays for on every turn.
4
+ */
5
+ import type { Workspace } from "./types.js";
6
+ /** Memory files a host agent loads automatically. */
7
+ export declare const HOST_FILES: string[];
8
+ /** Names that mean "this is the OnDemandMemory list for the directory". */
9
+ export declare const LIST_FILE_NAMES: string[];
10
+ /**
11
+ * Directories never descended into when looking for memory content.
12
+ * This governs traversal only. A directory named here is still audited when the user
13
+ * points at it explicitly, which is how an archived corpus gets inspected.
14
+ */
15
+ export declare const SKIP_DIRS: Set<string>;
16
+ export declare class DiscoveryError extends Error {
17
+ }
18
+ /** True for a regular file that is not a symlink. `lstat` on purpose: a cloned repo can ship a link. */
19
+ export declare function isPlainFile(abs: string): boolean;
20
+ /**
21
+ * Which tool surface a root can support, decided from directory entries alone.
22
+ *
23
+ * `createMcpServer` needs the shape of a root before it registers anything, on every launch.
24
+ * `discover()` cannot serve that: it reads the content of every memory file and folds in the
25
+ * auto-memory folder, which is most of the 310 to 380ms `doctor` cost the 2026-09-13 audit
26
+ * measured against a 68-file corpus. This probe does `stat` only, never a read, and tests the
27
+ * one distinction that changes the answer: `discover()` checks for `.minnimemory/` before it
28
+ * checks any other shape, so agreeing with it on a compiled root is the whole job. Every other
29
+ * shape it can return maps to the same "setup" surface.
30
+ *
31
+ * Every failure resolves to "setup": that surface holds the tools that diagnose and repair a
32
+ * root, so a root we cannot classify gets the tools that can tell the user why.
33
+ *
34
+ * Requires `manifest.json` inside `.minnimemory/` too (stat only, never read): a bare
35
+ * `.minnimemory/` directory with nothing compiled into it yet is `setup`, not `compiled`.
36
+ *
37
+ * `memory-dir` (added 2026-09-16, manifest-less recall): a root with no `.minnimemory/` at all
38
+ * that would still discover() to shape `memory-dir` or `auto-memory` - an index file (one of
39
+ * `LIST_FILE_NAMES`) plus topic files, three or more loose markdown files with no index, or a
40
+ * Claude Code auto-memory folder with content. Decided the same cheap way as the rest of this
41
+ * function: directory entries and a stat per candidate, never a file read, so this stays the
42
+ * probe discover() itself is too expensive to be. A root with a `.minnimemory/` directory (even
43
+ * an empty one, or one missing its manifest) never falls through to this check - `.minnimemory/`
44
+ * existing at all means "setup", exactly as before.
45
+ */
46
+ export type WorkspacePhase = "compiled" | "memory-dir" | "setup";
47
+ export declare function detectPhase(root: string): WorkspacePhase;
48
+ export interface ImportResolution {
49
+ /** absolute paths of imported files inside the workspace root */
50
+ files: string[];
51
+ /**
52
+ * Imports that resolved outside the root and were not read. Reported as the raw text the
53
+ * memory file wrote, never as a resolved absolute path: the point is to not disclose where
54
+ * on this machine the target would have been.
55
+ */
56
+ skipped: string[];
57
+ }
58
+ /**
59
+ * Resolve the files a memory file pulls into the always-loaded prefix through `@path` imports,
60
+ * the way Claude Code does: relative to the importing file, `~/` allowed, recursive to a depth
61
+ * of five, each file once. A path that does not resolve is ignored rather than reported: an
62
+ * `@` in prose is not an error.
63
+ *
64
+ * Imports that land outside `root` are NOT followed unless `followExternal` is set.
65
+ *
66
+ * The memory file being audited is frequently one you did not write - the whole pitch is
67
+ * pointing this at a repo, including in CI - and an `@` line is content that repo controls.
68
+ * Following `@~/.claude/CLAUDE.md` from a cloned repo read seventeen files out of the auditor's
69
+ * home directory, whose headings, frontmatter, body-derived keywords and credential prefixes
70
+ * then surfaced in doctor and scan output. This is the same threat findHostFile already refuses
71
+ * symlinks for ("a cloned repo can point a memory file at anything on your machine"); imports
72
+ * were the second door to the same room.
73
+ *
74
+ * Following them remains correct for your own repo, where an out-of-root import genuinely is
75
+ * part of the prefix you pay for, so it stays available - as a decision the person running the
76
+ * tool makes, not one the scanned repo makes for them.
77
+ */
78
+ export declare function resolveImports(hostAbs: string, root: string, followExternal?: boolean): ImportResolution;
79
+ /**
80
+ * A pointer stub is a host file whose entire content, headings and comments aside, names one
81
+ * other memory file: "@AGENTS.md", "AGENTS.md", "See `AGENTS.md` for the shared instructions".
82
+ * Eight of the twenty-three public CLAUDE.md files in the 2026-09-02 sweep were exactly this.
83
+ * Returns the absolute path of the named file when it exists next to the host, else undefined.
84
+ */
85
+ export declare function pointerTargetOf(hostAbs: string, content: string): string | undefined;
86
+ /**
87
+ * Claude Code's own project-key algorithm, reverse-engineered empirically (not published), and
88
+ * verified against this machine's real `~/.claude/projects/` folder names across a dozen
89
+ * independent examples, including nested repos and a path containing `&`: every character that
90
+ * is not a-z, A-Z, or 0-9 becomes a single `-`. No collapsing of adjacent replacements, no case
91
+ * change. `C:\Code` -> `C--Code`; `C:\Code\Widgets\App` -> `C--Code-Widgets-App`.
92
+ */
93
+ export declare function claudeProjectSlug(absPath: string): string;
94
+ /**
95
+ * Locate the auto-memory folder that belongs to `target`, the same way Claude Code itself would
96
+ * resolve it: git-repo-root if `target` is inside a repo, `target` itself otherwise, then
97
+ * Claude Code's own slug algorithm. Returns undefined, never throws, when nothing is found there -
98
+ * most targets simply have never been launched with Claude Code and that is not an error.
99
+ */
100
+ export declare function locateAutoMemoryDir(target: string): string | undefined;
101
+ export interface DiscoverOptions {
102
+ /**
103
+ * Follow `@path` imports that resolve outside the workspace root. Off by default: the file
104
+ * being audited is often one you did not write. See resolveImports.
105
+ */
106
+ followExternalImports?: boolean;
107
+ /**
108
+ * Fold in Claude Code's OS-level auto-memory folder (`~/.claude/projects/<slug>/memory/`),
109
+ * outside the target directory entirely. Default true - this is the two-location unification
110
+ * feature, and the CLI's own `doctor`/`bench` intentionally see it with no flag needed, since
111
+ * the operator running the CLI on their own machine and the memory it audits are the same
112
+ * party. Explicitly set false for the MCP server's scan/doctor tools: a target confined to
113
+ * "this server's root" per their own descriptions must not silently widen to the operator's
114
+ * entire global memory, the same category of reach `followExternalImports` already gates for
115
+ * `@path` imports (2026-09-10 audit).
116
+ */
117
+ includeAutoMemory?: boolean;
118
+ }
119
+ /**
120
+ * Inspect `target` and build a Workspace.
121
+ *
122
+ * The important output is which files carry `alwaysLoaded: true`, because that set is what
123
+ * the whole tool is trying to shrink.
124
+ */
125
+ export declare function discover(target: string, options?: DiscoverOptions): Workspace;