minnimemory 1.0.0-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +39 -0
- package/README.md +824 -0
- package/dist/bench.d.ts +98 -0
- package/dist/bench.js +142 -0
- package/dist/benchReport.d.ts +12 -0
- package/dist/benchReport.js +128 -0
- package/dist/bounds.d.ts +40 -0
- package/dist/bounds.js +44 -0
- package/dist/cli.d.ts +15 -0
- package/dist/cli.js +503 -0
- package/dist/compile.d.ts +187 -0
- package/dist/compile.js +516 -0
- package/dist/discover.d.ts +125 -0
- package/dist/discover.js +520 -0
- package/dist/doctor.d.ts +9 -0
- package/dist/doctor.js +67 -0
- package/dist/episodic.d.ts +47 -0
- package/dist/episodic.js +130 -0
- package/dist/hook.d.ts +45 -0
- package/dist/hook.js +104 -0
- package/dist/index.d.ts +18 -0
- package/dist/index.js +18 -0
- package/dist/init.d.ts +125 -0
- package/dist/init.js +475 -0
- package/dist/instructions.d.ts +60 -0
- package/dist/instructions.js +270 -0
- package/dist/mcp.d.ts +109 -0
- package/dist/mcp.js +252 -0
- package/dist/mcpServer.d.ts +136 -0
- package/dist/mcpServer.js +997 -0
- package/dist/paths.d.ts +25 -0
- package/dist/paths.js +47 -0
- package/dist/recall.d.ts +113 -0
- package/dist/recall.js +256 -0
- package/dist/recallDir.d.ts +50 -0
- package/dist/recallDir.js +187 -0
- package/dist/reorganize.d.ts +62 -0
- package/dist/reorganize.js +216 -0
- package/dist/report.d.ts +16 -0
- package/dist/report.js +204 -0
- package/dist/router.d.ts +141 -0
- package/dist/router.js +314 -0
- package/dist/rules.d.ts +32 -0
- package/dist/rules.js +651 -0
- package/dist/scan.d.ts +110 -0
- package/dist/scan.js +173 -0
- package/dist/text.d.ts +158 -0
- package/dist/text.js +395 -0
- package/dist/tokenizer.d.ts +26 -0
- package/dist/tokenizer.js +69 -0
- package/dist/types.d.ts +156 -0
- package/dist/types.js +17 -0
- package/dist/version.d.ts +7 -0
- package/dist/version.js +7 -0
- package/dist/writeProtocol.d.ts +19 -0
- package/dist/writeProtocol.js +45 -0
- package/examples/CLAUDE.md +75 -0
- package/examples/README.md +7 -0
- package/package.json +52 -0
package/dist/compile.js
ADDED
|
@@ -0,0 +1,516 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The compiler: turn one memory file into a cache-stable AlwaysOnMemory body plus routed
|
|
3
|
+
* OnDemandMemory files.
|
|
4
|
+
*
|
|
5
|
+
* Pure by design. Everything here is string in, strings out, so the whole compile can be
|
|
6
|
+
* tested from fixtures and diffed before anything touches a user's disk.
|
|
7
|
+
*
|
|
8
|
+
* Structure follows the tiered-memory pattern already running in production in the Minni
|
|
9
|
+
* agents (AlwaysOnMemory.md, OnDemandMemory/, fixed assembly order), and the emitted
|
|
10
|
+
* instruction block comes from the MinniMemory v1.0 research set. See instructions.ts.
|
|
11
|
+
*/
|
|
12
|
+
import { createHash } from "node:crypto";
|
|
13
|
+
import { defaultProfile, profile as profileFor, renderInstructions } from "./instructions.js";
|
|
14
|
+
import { classifyKind, episodicFromMarkdown, episodicToJson } from "./episodic.js";
|
|
15
|
+
import { renderWriteProtocolFile, WRITE_PROTOCOL_HEADING, WRITE_PROTOCOL_FILE_NAME, WRITE_PROTOCOL_TRIGGERS, } from "./writeProtocol.js";
|
|
16
|
+
import { blocks, GRANULARITY_TOKENS, LIST_END, LIST_START, INSTRUCTIONS_END, INSTRUCTIONS_START, keywords, sections, sectionVolatilityReasons, slugify, splitFrontmatter, volatilityReasons, } from "./text.js";
|
|
17
|
+
import { selectTriggers } from "./router.js";
|
|
18
|
+
import { estimateTokens } from "./tokenizer.js";
|
|
19
|
+
/** Marker identifying a host file this tool generated, so it is never double counted. */
|
|
20
|
+
export const STUB_MARKER = "<!-- minnimemory:stub v1 -->";
|
|
21
|
+
/** Default always-loaded budget in tokens; matches doctor's MM001 default. */
|
|
22
|
+
export const DEFAULT_BUDGET = 2000;
|
|
23
|
+
/**
|
|
24
|
+
* Rough cost of one OnDemandMemory list line and of the fixed list scaffolding (markers,
|
|
25
|
+
* heading, and the routing line the `none` profile adds), used only to keep AlwaysOnMemory
|
|
26
|
+
* placement inside the budget. Measured with approx-v2 on the list-form OnDemandMemory list
|
|
27
|
+
* (2026-09-04 audit); the previous table form cost 159 fixed while the estimator assumed 90.
|
|
28
|
+
*/
|
|
29
|
+
const LIST_LINE_COST = 22;
|
|
30
|
+
const LIST_BASE_COST = 40;
|
|
31
|
+
/** Headings whose content is identity or invariant, so it belongs in AlwaysOnMemory. */
|
|
32
|
+
export const ALWAYS_ON_HINTS = /\b(identity|overview|about|purpose|mission|principles?|conventions?|rules?|constraints?|boundar\w*|guardrails?|guidelines?|style|tone|standards?|polic(?:y|ies)|invariants?|philosophy|decisions?)\b/i;
|
|
33
|
+
/**
|
|
34
|
+
* Below this, an OnDemandMemory file is not worth routing to: the list line and the decision
|
|
35
|
+
* to load it cost more than just having read the content. Consecutive small sections are merged.
|
|
36
|
+
*/
|
|
37
|
+
const MIN_ON_DEMAND_TOKENS = 400;
|
|
38
|
+
function hash(text) {
|
|
39
|
+
return createHash("sha256").update(text).digest("hex").slice(0, 16);
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* The hash MM010 and recall's drift check both compare against a manifest entry.
|
|
43
|
+
*
|
|
44
|
+
* CRLF-normalised and trailing-whitespace-trimmed on purpose: a checkout can rewrite line
|
|
45
|
+
* endings with nobody editing anything, and a file that differs only that way has not drifted.
|
|
46
|
+
* One copy, because the dated-bullet regex taught us what three copies of a rule costs
|
|
47
|
+
* (2026-09-13 audit, D7).
|
|
48
|
+
*/
|
|
49
|
+
export function driftHash(content) {
|
|
50
|
+
return hash(content.replace(/\r\n/g, "\n").trimEnd());
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* Shared with rules.ts's MM010 check against AlwaysOnMemory.md and the stub: sha256 slice 16
|
|
54
|
+
* with no normalisation, the caller's job (CRLF-only, no trimEnd - those files are written
|
|
55
|
+
* verbatim, so trimming would blur an edit driftHash would forgive on purpose for OnDemandMemory
|
|
56
|
+
* files but should not here).
|
|
57
|
+
*/
|
|
58
|
+
export { hash as verbatimHash };
|
|
59
|
+
/** Pick the heading level that actually divides the document into topics. */
|
|
60
|
+
function detectSplitLevel(all) {
|
|
61
|
+
for (let level = 2; level <= 4; level++) {
|
|
62
|
+
if (all.filter((s) => s.level === level).length >= 2)
|
|
63
|
+
return level;
|
|
64
|
+
}
|
|
65
|
+
// A document of only H1s still needs splitting somewhere.
|
|
66
|
+
if (all.filter((s) => s.level === 1).length >= 2)
|
|
67
|
+
return 1;
|
|
68
|
+
return 2;
|
|
69
|
+
}
|
|
70
|
+
function textOf(lines, startLine, endLine) {
|
|
71
|
+
return lines.slice(startLine - 1, endLine).join("\n").trimEnd();
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Choose each OnDemandMemory file's list triggers against the whole file set (router.ts
|
|
75
|
+
* selectTriggers): frequency here times rarity elsewhere, headings weighted. Replaces the
|
|
76
|
+
* frequency-only keywords() pick plus a post-hoc prune, which produced `bars, free, merged` for
|
|
77
|
+
* a trading-app memory file (audit 2026-09-04). The per-file keywords() call remains the
|
|
78
|
+
* fallback for paths that see one file at a time (init --update).
|
|
79
|
+
*/
|
|
80
|
+
function assignTriggers(onDemandFiles) {
|
|
81
|
+
const chosen = selectTriggers(onDemandFiles.map((m) => ({ name: m.name, heading: m.heading, content: m.content })));
|
|
82
|
+
for (const m of onDemandFiles)
|
|
83
|
+
m.triggers = chosen.get(m.name) ?? m.triggers;
|
|
84
|
+
}
|
|
85
|
+
/**
|
|
86
|
+
* Merge runs of small consecutive sections into OnDemandMemory files worth routing to.
|
|
87
|
+
* Order is preserved, so the fixed assembly order still matches document order.
|
|
88
|
+
*/
|
|
89
|
+
function mergeSmall(onDemandFiles, min) {
|
|
90
|
+
const out = [];
|
|
91
|
+
let bucket = [];
|
|
92
|
+
const flush = () => {
|
|
93
|
+
if (bucket.length === 0)
|
|
94
|
+
return;
|
|
95
|
+
const first = bucket[0];
|
|
96
|
+
if (!first) {
|
|
97
|
+
bucket = [];
|
|
98
|
+
return;
|
|
99
|
+
}
|
|
100
|
+
if (bucket.length === 1) {
|
|
101
|
+
out.push(first);
|
|
102
|
+
bucket = [];
|
|
103
|
+
return;
|
|
104
|
+
}
|
|
105
|
+
const last = bucket[bucket.length - 1] ?? first;
|
|
106
|
+
const content = bucket.map((b) => b.content).join("\n\n");
|
|
107
|
+
const triggers = [...new Set(bucket.flatMap((b) => b.triggers))].slice(0, 8);
|
|
108
|
+
// A merged OnDemandMemory file must not inherit the first section's name: a file called
|
|
109
|
+
// current_status.md that also holds testing and deployment misleads the reader
|
|
110
|
+
// and the agent about what is inside it.
|
|
111
|
+
const name = slugify(`${first.heading} to ${last.heading}`);
|
|
112
|
+
out.push({
|
|
113
|
+
name,
|
|
114
|
+
file: `${ON_DEMAND_DIR_NAME}/${name}.md`,
|
|
115
|
+
heading: `${first.heading} through ${last.heading}`,
|
|
116
|
+
content,
|
|
117
|
+
tokens: estimateTokens(content),
|
|
118
|
+
triggers,
|
|
119
|
+
sourceStartLine: first.sourceStartLine,
|
|
120
|
+
sourceEndLine: last.sourceEndLine,
|
|
121
|
+
reason: bucket.some((b) => b.reason === "volatile")
|
|
122
|
+
? "volatile"
|
|
123
|
+
: bucket.some((b) => b.reason === "over-budget")
|
|
124
|
+
? "over-budget"
|
|
125
|
+
: "task-specific",
|
|
126
|
+
kind: first.kind,
|
|
127
|
+
});
|
|
128
|
+
bucket = [];
|
|
129
|
+
};
|
|
130
|
+
for (const m of onDemandFiles) {
|
|
131
|
+
// A file the granularity split produced stands alone: flush whatever was accumulating,
|
|
132
|
+
// emit it untouched, and start a fresh bucket after it. So does an episodic section (O2):
|
|
133
|
+
// it is append-only and may become JSON (O3), so it never shares a file with prose.
|
|
134
|
+
if (m.split || m.kind === "episodic") {
|
|
135
|
+
flush();
|
|
136
|
+
out.push(m);
|
|
137
|
+
continue;
|
|
138
|
+
}
|
|
139
|
+
// Kinds are edited differently (O2), so a bucket holds one kind only.
|
|
140
|
+
if (bucket.length > 0 && bucket[0]?.kind !== m.kind)
|
|
141
|
+
flush();
|
|
142
|
+
bucket.push(m);
|
|
143
|
+
if (bucket.reduce((n, b) => n + b.tokens, 0) >= min)
|
|
144
|
+
flush();
|
|
145
|
+
}
|
|
146
|
+
flush();
|
|
147
|
+
return out;
|
|
148
|
+
}
|
|
149
|
+
/**
|
|
150
|
+
* The list title is deliberately not derived from the document title: a title can carry a
|
|
151
|
+
* date or a status word ("Minni (renamed 2026-08-11)"), and the list must stay free of anything
|
|
152
|
+
* the volatility rules would flag. The whole list is fenced with markers so `doctor` can tell a
|
|
153
|
+
* routing manifest from prose.
|
|
154
|
+
*
|
|
155
|
+
* One line per OnDemandMemory file and nothing that restates a Part 2 rule (P4). Rules 15, 16,
|
|
156
|
+
* 22 and 23 already say how to use the list and sit in the same prefix whenever a discipline
|
|
157
|
+
* block is embedded; the previous table-plus-prose form repeated them at 159 tokens per compile.
|
|
158
|
+
* Only the `none` profile, which embeds no block, gets one routing line after the list. Assembly
|
|
159
|
+
* order is the manifest's `order`, not prefix text.
|
|
160
|
+
*/
|
|
161
|
+
/** The directory `init` writes next to the host file. Named here, where the stub's
|
|
162
|
+
* OnDemandMemory list is rendered, so the two can never disagree about where a file lives. */
|
|
163
|
+
export const COMPILED_DIR_NAME = ".minnimemory";
|
|
164
|
+
/** The one always-loaded file inside the compiled directory: AlwaysOnMemory body plus the
|
|
165
|
+
* OnDemandMemory list. */
|
|
166
|
+
export const ALWAYS_ON_FILE_NAME = "AlwaysOnMemory.md";
|
|
167
|
+
/** The OnDemandMemory files directory inside the compiled directory. */
|
|
168
|
+
export const ON_DEMAND_DIR_NAME = "OnDemandMemory";
|
|
169
|
+
function renderOnDemandList(onDemandFiles, withRoutingLine, pathPrefix = "") {
|
|
170
|
+
const lines = [LIST_START, "# OnDemandMemory"];
|
|
171
|
+
if (onDemandFiles.length === 0) {
|
|
172
|
+
lines.push("No OnDemandMemory files. Everything lives in AlwaysOnMemory.", LIST_END, "");
|
|
173
|
+
return lines.join("\n");
|
|
174
|
+
}
|
|
175
|
+
// Paths are relative to the file the list sits in. AlwaysOnMemory.md lives inside the
|
|
176
|
+
// compiled directory beside OnDemandMemory/; the stub lives one level up, in the project, and
|
|
177
|
+
// an agent that reads `OnDemandMemory/x.md` from there finds nothing. The 2026-09-05 session
|
|
178
|
+
// measurement caught exactly that: the agent tried the bare path, missed, and gave up.
|
|
179
|
+
for (const m of onDemandFiles) {
|
|
180
|
+
lines.push(`- \`${pathPrefix}${m.file}\`: ${m.triggers.join(", ")}`);
|
|
181
|
+
}
|
|
182
|
+
if (withRoutingLine) {
|
|
183
|
+
lines.push("", "Read an OnDemandMemory file only when the task needs it, and answer from it, never from this list.");
|
|
184
|
+
}
|
|
185
|
+
lines.push(LIST_END, "");
|
|
186
|
+
return lines.join("\n");
|
|
187
|
+
}
|
|
188
|
+
/** The instruction block, fenced so `doctor` skips it and `init --update` can strip it. */
|
|
189
|
+
function renderInstructionRegion(profile) {
|
|
190
|
+
const body = renderInstructions(profile);
|
|
191
|
+
if (!body)
|
|
192
|
+
return "";
|
|
193
|
+
return [INSTRUCTIONS_START, body.trimEnd(), INSTRUCTIONS_END].join("\n");
|
|
194
|
+
}
|
|
195
|
+
function renderAlwaysOnBody(frontmatter, title, preamble, alwaysOnSections, profile) {
|
|
196
|
+
const parts = [];
|
|
197
|
+
if (frontmatter)
|
|
198
|
+
parts.push(frontmatter, "");
|
|
199
|
+
parts.push(`# ${title}`, "");
|
|
200
|
+
if (preamble.trim())
|
|
201
|
+
parts.push(preamble.trim(), "");
|
|
202
|
+
for (const s of alwaysOnSections)
|
|
203
|
+
parts.push(s.trim(), "");
|
|
204
|
+
const instructions = renderInstructionRegion(profile);
|
|
205
|
+
if (instructions)
|
|
206
|
+
parts.push(instructions, "");
|
|
207
|
+
return parts.join("\n");
|
|
208
|
+
}
|
|
209
|
+
/**
|
|
210
|
+
* Frontmatter, when present, stays the very first thing in the stub: a YAML header that is not
|
|
211
|
+
* on line 1 stops being a header. The generated-file marker follows it. `alwaysOnBody` here is
|
|
212
|
+
* the AlwaysOnMemory body without its frontmatter, since the stub places the frontmatter itself.
|
|
213
|
+
*/
|
|
214
|
+
function renderStub(frontmatter, alwaysOnBody, onDemandList, onDemandCount) {
|
|
215
|
+
const parts = [];
|
|
216
|
+
if (frontmatter)
|
|
217
|
+
parts.push(frontmatter);
|
|
218
|
+
parts.push(STUB_MARKER, "<!-- generated by minnimemory; edit .minnimemory/ instead -->", "");
|
|
219
|
+
parts.push(alwaysOnBody.trimEnd(), "");
|
|
220
|
+
if (onDemandCount > 0)
|
|
221
|
+
parts.push(onDemandList.trimEnd(), "");
|
|
222
|
+
return parts.join("\n");
|
|
223
|
+
}
|
|
224
|
+
/**
|
|
225
|
+
* `.minnimemory/AlwaysOnMemory.md`: the AlwaysOnMemory body (already carries its own
|
|
226
|
+
* frontmatter) plus the OnDemandMemory list, the same join the stub uses, written to its own
|
|
227
|
+
* file instead of into the host file.
|
|
228
|
+
*/
|
|
229
|
+
function renderAlwaysOn(alwaysOnBody, onDemandList, onDemandCount) {
|
|
230
|
+
const parts = [alwaysOnBody.trimEnd(), ""];
|
|
231
|
+
if (onDemandCount > 0)
|
|
232
|
+
parts.push(onDemandList.trimEnd(), "");
|
|
233
|
+
return parts.join("\n");
|
|
234
|
+
}
|
|
235
|
+
/**
|
|
236
|
+
* Is this host file a stub this tool generated? Only a marker on the first line after any
|
|
237
|
+
* frontmatter counts. A marker quoted in a code block, or anywhere else in prose, is text
|
|
238
|
+
* (security audit 2026-09-02, finding 5).
|
|
239
|
+
*/
|
|
240
|
+
export function isStub(source) {
|
|
241
|
+
const lines = source.split(/\r?\n/);
|
|
242
|
+
const { bodyStartLine } = splitFrontmatter(lines);
|
|
243
|
+
return (lines[bodyStartLine - 1] ?? "").trim() === STUB_MARKER;
|
|
244
|
+
}
|
|
245
|
+
export function compile(source, options) {
|
|
246
|
+
const lines = source.split(/\r?\n/);
|
|
247
|
+
const { frontmatter: frontmatterLines, bodyStartLine } = splitFrontmatter(lines);
|
|
248
|
+
const frontmatter = frontmatterLines.join("\n");
|
|
249
|
+
const all = sections(lines).filter((s) => s.startLine >= bodyStartLine);
|
|
250
|
+
const profile = options.profile ?? (options.profileName ? profileFor(options.profileName) : defaultProfile());
|
|
251
|
+
const splitLevel = options.splitLevel ?? detectSplitLevel(all);
|
|
252
|
+
const splits = all.filter((s) => s.level === splitLevel);
|
|
253
|
+
const firstSplitLine = splits[0]?.startLine ?? lines.length + 1;
|
|
254
|
+
// The title is an H1 that comes before the first split heading. An H1 further down is a
|
|
255
|
+
// section like any other, not the document's name.
|
|
256
|
+
const titleSection = all.find((s) => s.level === 1 && s.startLine < firstSplitLine);
|
|
257
|
+
const title = titleSection?.heading ?? options.sourceName.replace(/\.md$/i, "");
|
|
258
|
+
// Everything above the first split heading is preamble. That includes any prose that sits
|
|
259
|
+
// above the title line, so nothing written before the H1 is lost.
|
|
260
|
+
const preambleStart = bodyStartLine;
|
|
261
|
+
const preambleLines = titleSection
|
|
262
|
+
? [
|
|
263
|
+
...lines.slice(bodyStartLine - 1, titleSection.startLine - 1),
|
|
264
|
+
...lines.slice(titleSection.startLine, firstSplitLine - 1),
|
|
265
|
+
]
|
|
266
|
+
: lines.slice(bodyStartLine - 1, firstSplitLine - 1);
|
|
267
|
+
const preamble = preambleLines.join("\n").trimEnd();
|
|
268
|
+
const alwaysOnSections = [];
|
|
269
|
+
const onDemandFiles = [];
|
|
270
|
+
// The preamble is not exempt from the cache-stability law. Split it per block and route the
|
|
271
|
+
// volatile blocks out, so a dated audit note cannot sit in the always-loaded prefix just
|
|
272
|
+
// because it happened to appear above the first heading.
|
|
273
|
+
const preambleBlocks = blocks(preamble.split("\n"));
|
|
274
|
+
const stablePreamble = preambleBlocks.filter((b) => volatilityReasons(b.text).length === 0);
|
|
275
|
+
const volatilePreamble = preambleBlocks.filter((b) => volatilityReasons(b.text).length > 0);
|
|
276
|
+
if (volatilePreamble.length > 0) {
|
|
277
|
+
const content = ["## Notes", "", ...volatilePreamble.map((b) => b.text)].join("\n\n");
|
|
278
|
+
onDemandFiles.push({
|
|
279
|
+
name: "notes",
|
|
280
|
+
file: `${ON_DEMAND_DIR_NAME}/notes.md`,
|
|
281
|
+
heading: "Notes",
|
|
282
|
+
content,
|
|
283
|
+
tokens: estimateTokens(content),
|
|
284
|
+
triggers: keywords("notes status", content),
|
|
285
|
+
sourceStartLine: preambleStart,
|
|
286
|
+
sourceEndLine: firstSplitLine - 1,
|
|
287
|
+
reason: "volatile",
|
|
288
|
+
});
|
|
289
|
+
}
|
|
290
|
+
const preambleText = stablePreamble.map((b) => b.text).join("\n\n");
|
|
291
|
+
const entries = splits.map((s, i) => {
|
|
292
|
+
const next = splits[i + 1];
|
|
293
|
+
const endLine = next ? next.startLine - 1 : lines.length;
|
|
294
|
+
const content = textOf(lines, s.startLine, endLine);
|
|
295
|
+
const volatile = sectionVolatilityReasons(content.split("\n"));
|
|
296
|
+
// The cache-stability law: volatile content never enters the always-loaded prefix,
|
|
297
|
+
// no matter how always-on-like its heading reads.
|
|
298
|
+
const alwaysOnCandidate = ALWAYS_ON_HINTS.test(s.heading) && volatile.length === 0;
|
|
299
|
+
return { section: s, content, endLine, tokens: estimateTokens(content), volatile, alwaysOnCandidate };
|
|
300
|
+
});
|
|
301
|
+
// The stable preamble competes for AlwaysOnMemory like any other candidate. Real AGENTS.md
|
|
302
|
+
// files put a thousand tokens of prose before the first H2; exempting that from the budget
|
|
303
|
+
// would push every actual rule section out instead.
|
|
304
|
+
if (preambleText.trim()) {
|
|
305
|
+
entries.unshift({
|
|
306
|
+
section: { heading: title, level: 0, startLine: preambleStart, endLine: firstSplitLine - 1 },
|
|
307
|
+
content: preambleText,
|
|
308
|
+
endLine: firstSplitLine - 1,
|
|
309
|
+
tokens: estimateTokens(preambleText),
|
|
310
|
+
volatile: [],
|
|
311
|
+
alwaysOnCandidate: true,
|
|
312
|
+
preamble: true,
|
|
313
|
+
});
|
|
314
|
+
}
|
|
315
|
+
// Budget-aware placement. A heading can read as always-on ("Architecture guidelines", "Code
|
|
316
|
+
// Style") and still be thousands of tokens; keeping every such section in AlwaysOnMemory just
|
|
317
|
+
// moves the MM001 finding from the source to the compiled output. Smallest candidates are kept
|
|
318
|
+
// first, because a short set of invariants is what AlwaysOnMemory is for, and whatever does
|
|
319
|
+
// not fit is routed to an OnDemandMemory file with the reason recorded so the user can see it
|
|
320
|
+
// and raise --budget.
|
|
321
|
+
const budget = options.budget ?? DEFAULT_BUDGET;
|
|
322
|
+
const candidates = entries.filter((e) => e.alwaysOnCandidate).sort((a, b) => a.tokens - b.tokens);
|
|
323
|
+
const onDemandEstimate = entries.length - candidates.length + (volatilePreamble.length > 0 ? 1 : 0);
|
|
324
|
+
let used = estimateTokens(renderAlwaysOnBody("", title, "", [], profile)) + LIST_BASE_COST + LIST_LINE_COST * onDemandEstimate;
|
|
325
|
+
const kept = new Set();
|
|
326
|
+
for (const c of candidates) {
|
|
327
|
+
if (used + c.tokens <= budget) {
|
|
328
|
+
kept.add(c);
|
|
329
|
+
used += c.tokens;
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
const demoted = candidates.filter((c) => !kept.has(c));
|
|
333
|
+
const stableText = entries.find((e) => e.preamble && kept.has(e))?.content ?? "";
|
|
334
|
+
for (const e of entries) {
|
|
335
|
+
if (kept.has(e)) {
|
|
336
|
+
if (!e.preamble)
|
|
337
|
+
alwaysOnSections.push(e.content);
|
|
338
|
+
continue;
|
|
339
|
+
}
|
|
340
|
+
if (e.preamble) {
|
|
341
|
+
const name = slugify(`${title} overview`);
|
|
342
|
+
onDemandFiles.push({
|
|
343
|
+
name,
|
|
344
|
+
file: `${ON_DEMAND_DIR_NAME}/${name}.md`,
|
|
345
|
+
heading: `${title} overview`,
|
|
346
|
+
content: `## ${title} overview\n\n${e.content}`,
|
|
347
|
+
tokens: e.tokens,
|
|
348
|
+
triggers: keywords(`${title} overview`, e.content),
|
|
349
|
+
sourceStartLine: e.section.startLine,
|
|
350
|
+
sourceEndLine: e.endLine,
|
|
351
|
+
reason: "over-budget",
|
|
352
|
+
});
|
|
353
|
+
continue;
|
|
354
|
+
}
|
|
355
|
+
const reason = e.volatile.length > 0 ? "volatile" : e.alwaysOnCandidate ? "over-budget" : "task-specific";
|
|
356
|
+
// The granularity clause (O7) on the agent-driven route: an agent that reads a whole file
|
|
357
|
+
// the OnDemandMemory list named pays the whole file, so a section over the threshold is
|
|
358
|
+
// emitted as one OnDemandMemory file per subheading, the parent heading and its intro lines
|
|
359
|
+
// first. Every line lands in exactly one file, in document order, so losslessness and the
|
|
360
|
+
// assembly order both hold. A flat section has nothing to split on; doctor's MM009 says so.
|
|
361
|
+
const subs = e.tokens >= GRANULARITY_TOKENS
|
|
362
|
+
? all.filter((s) => s.level === splitLevel + 1 && s.startLine > e.section.startLine && s.startLine <= e.endLine)
|
|
363
|
+
: [];
|
|
364
|
+
if (subs.length > 0) {
|
|
365
|
+
const first = subs[0];
|
|
366
|
+
const introEnd = first.startLine - 1;
|
|
367
|
+
const intro = textOf(lines, e.section.startLine, introEnd);
|
|
368
|
+
const introTokens = estimateTokens(intro);
|
|
369
|
+
// A parent heading plus a sentence or two is not worth its own list line, so it rides
|
|
370
|
+
// with the first subheading instead. Only a substantial intro becomes its own file.
|
|
371
|
+
const introStandsAlone = introTokens >= LIST_LINE_COST * 2;
|
|
372
|
+
if (introStandsAlone) {
|
|
373
|
+
onDemandFiles.push({
|
|
374
|
+
name: slugify(e.section.heading),
|
|
375
|
+
file: `${ON_DEMAND_DIR_NAME}/${slugify(e.section.heading)}.md`,
|
|
376
|
+
heading: e.section.heading,
|
|
377
|
+
content: intro,
|
|
378
|
+
tokens: introTokens,
|
|
379
|
+
triggers: keywords(e.section.heading, intro),
|
|
380
|
+
sourceStartLine: e.section.startLine,
|
|
381
|
+
sourceEndLine: introEnd,
|
|
382
|
+
reason,
|
|
383
|
+
split: true,
|
|
384
|
+
});
|
|
385
|
+
}
|
|
386
|
+
subs.forEach((sub, i) => {
|
|
387
|
+
const next = subs[i + 1];
|
|
388
|
+
const end = next ? next.startLine - 1 : e.endLine;
|
|
389
|
+
const isFirst = i === 0;
|
|
390
|
+
const startLine = !introStandsAlone && isFirst ? e.section.startLine : sub.startLine;
|
|
391
|
+
const content = textOf(lines, startLine, end);
|
|
392
|
+
const heading = `${e.section.heading}: ${sub.heading}`;
|
|
393
|
+
const name = slugify(heading);
|
|
394
|
+
onDemandFiles.push({
|
|
395
|
+
name,
|
|
396
|
+
file: `${ON_DEMAND_DIR_NAME}/${name}.md`,
|
|
397
|
+
heading,
|
|
398
|
+
content,
|
|
399
|
+
tokens: estimateTokens(content),
|
|
400
|
+
triggers: keywords(heading, content),
|
|
401
|
+
sourceStartLine: startLine,
|
|
402
|
+
sourceEndLine: end,
|
|
403
|
+
reason,
|
|
404
|
+
split: true,
|
|
405
|
+
});
|
|
406
|
+
});
|
|
407
|
+
continue;
|
|
408
|
+
}
|
|
409
|
+
const name = slugify(e.section.heading);
|
|
410
|
+
onDemandFiles.push({
|
|
411
|
+
name,
|
|
412
|
+
file: `${ON_DEMAND_DIR_NAME}/${name}.md`,
|
|
413
|
+
heading: e.section.heading,
|
|
414
|
+
content: e.content,
|
|
415
|
+
tokens: e.tokens,
|
|
416
|
+
triggers: keywords(e.section.heading, e.content),
|
|
417
|
+
sourceStartLine: e.section.startLine,
|
|
418
|
+
sourceEndLine: e.endLine,
|
|
419
|
+
reason,
|
|
420
|
+
});
|
|
421
|
+
}
|
|
422
|
+
// O2 before merging: the kind decides what may share a file.
|
|
423
|
+
for (const m of onDemandFiles)
|
|
424
|
+
m.kind = m.kind ?? classifyKind(m.heading, m.content.split("\n").slice(1));
|
|
425
|
+
const merged = options.onDemandFilesOverride ?? mergeSmall(onDemandFiles, MIN_ON_DEMAND_TOKENS);
|
|
426
|
+
// Slugs must be unique or OnDemandMemory files overwrite each other on disk.
|
|
427
|
+
const seen = new Map();
|
|
428
|
+
for (const m of merged) {
|
|
429
|
+
const n = seen.get(m.name) ?? 0;
|
|
430
|
+
seen.set(m.name, n + 1);
|
|
431
|
+
if (n > 0) {
|
|
432
|
+
m.name = `${m.name}_${n + 1}`;
|
|
433
|
+
m.file = `${ON_DEMAND_DIR_NAME}/${m.name}.md`;
|
|
434
|
+
}
|
|
435
|
+
}
|
|
436
|
+
assignTriggers(merged);
|
|
437
|
+
// O2: every OnDemandMemory file carries its kind. O3: an episodic file is written as JSON
|
|
438
|
+
// when asked, and its on-disk bytes are what gets hashed and counted, so tokens follow the JSON.
|
|
439
|
+
for (const m of merged) {
|
|
440
|
+
if (m.generated)
|
|
441
|
+
continue;
|
|
442
|
+
m.kind = m.kind ?? classifyKind(m.heading, m.content.split("\n").slice(1));
|
|
443
|
+
if (options.episodicJson && m.kind === "episodic" && m.file.endsWith(".md")) {
|
|
444
|
+
m.content = episodicToJson(episodicFromMarkdown(m.content)).trimEnd();
|
|
445
|
+
m.file = m.file.replace(/\.md$/, ".json");
|
|
446
|
+
m.tokens = estimateTokens(m.content);
|
|
447
|
+
}
|
|
448
|
+
}
|
|
449
|
+
// O9: the write protocol rides as a routed OnDemandMemory file, never in the prefix. Appended
|
|
450
|
+
// last so the source OnDemandMemory files keep document order, and skipped when an override
|
|
451
|
+
// already has it.
|
|
452
|
+
if (options.writeProtocol !== false && !merged.some((m) => m.name === WRITE_PROTOCOL_FILE_NAME)) {
|
|
453
|
+
const content = renderWriteProtocolFile();
|
|
454
|
+
merged.push({
|
|
455
|
+
name: WRITE_PROTOCOL_FILE_NAME,
|
|
456
|
+
file: `${ON_DEMAND_DIR_NAME}/${WRITE_PROTOCOL_FILE_NAME}.md`,
|
|
457
|
+
heading: WRITE_PROTOCOL_HEADING,
|
|
458
|
+
content,
|
|
459
|
+
tokens: estimateTokens(content),
|
|
460
|
+
triggers: [...WRITE_PROTOCOL_TRIGGERS],
|
|
461
|
+
sourceStartLine: 0,
|
|
462
|
+
sourceEndLine: 0,
|
|
463
|
+
reason: "generated",
|
|
464
|
+
kind: "procedural",
|
|
465
|
+
generated: true,
|
|
466
|
+
});
|
|
467
|
+
}
|
|
468
|
+
const stubBody = renderAlwaysOnBody("", title, stableText, alwaysOnSections, profile);
|
|
469
|
+
const alwaysOnBody = renderAlwaysOnBody(frontmatter, title, stableText, alwaysOnSections, profile);
|
|
470
|
+
const onDemandList = renderOnDemandList(merged, profile.length === 0);
|
|
471
|
+
const stubOnDemandList = renderOnDemandList(merged, profile.length === 0, `${COMPILED_DIR_NAME}/`);
|
|
472
|
+
const stub = renderStub(frontmatter, stubBody, stubOnDemandList, merged.length);
|
|
473
|
+
const alwaysOn = renderAlwaysOn(alwaysOnBody, onDemandList, merged.length);
|
|
474
|
+
return {
|
|
475
|
+
title,
|
|
476
|
+
sourceName: options.sourceName,
|
|
477
|
+
sourceHash: hash(source),
|
|
478
|
+
frontmatter,
|
|
479
|
+
alwaysOnBody,
|
|
480
|
+
onDemandList,
|
|
481
|
+
alwaysOn,
|
|
482
|
+
stub,
|
|
483
|
+
onDemandFiles: merged,
|
|
484
|
+
before: estimateTokens(source),
|
|
485
|
+
after: estimateTokens(stub),
|
|
486
|
+
instructionTokens: estimateTokens(renderInstructions(profile)),
|
|
487
|
+
budget,
|
|
488
|
+
demoted: demoted
|
|
489
|
+
.sort((a, b) => a.section.startLine - b.section.startLine)
|
|
490
|
+
.map((d) => ({ heading: d.section.heading, tokens: d.tokens })),
|
|
491
|
+
};
|
|
492
|
+
}
|
|
493
|
+
export function buildManifest(compiled, tokenizer) {
|
|
494
|
+
return {
|
|
495
|
+
version: 4,
|
|
496
|
+
tokenizer,
|
|
497
|
+
generatedFrom: compiled.sourceName,
|
|
498
|
+
sourceHash: compiled.sourceHash,
|
|
499
|
+
instructionStatus: "output-side rules measured 2026-09-05; context-side rules unmeasured; see the Measurement policy section of the MinniMemoryMCP README",
|
|
500
|
+
order: compiled.onDemandFiles.map((m) => m.name),
|
|
501
|
+
prefix: { before: compiled.before, after: compiled.after },
|
|
502
|
+
always: { file: ALWAYS_ON_FILE_NAME, tokens: estimateTokens(compiled.alwaysOn), hash: hash(compiled.alwaysOn) },
|
|
503
|
+
stub: { file: compiled.sourceName, tokens: estimateTokens(compiled.stub), hash: hash(compiled.stub) },
|
|
504
|
+
onDemandFiles: compiled.onDemandFiles.map((m) => ({
|
|
505
|
+
name: m.name,
|
|
506
|
+
file: m.file,
|
|
507
|
+
tokens: m.tokens,
|
|
508
|
+
hash: driftHash(m.content),
|
|
509
|
+
triggers: m.triggers,
|
|
510
|
+
sourceLines: [m.sourceStartLine, m.sourceEndLine],
|
|
511
|
+
reason: m.reason,
|
|
512
|
+
kind: m.kind,
|
|
513
|
+
generated: m.generated ?? false,
|
|
514
|
+
})),
|
|
515
|
+
};
|
|
516
|
+
}
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Work out what kind of memory setup we are pointed at, and which files the model
|
|
3
|
+
* actually pays for on every turn.
|
|
4
|
+
*/
|
|
5
|
+
import type { Workspace } from "./types.js";
|
|
6
|
+
/** Memory files a host agent loads automatically. */
|
|
7
|
+
export declare const HOST_FILES: string[];
|
|
8
|
+
/** Names that mean "this is the OnDemandMemory list for the directory". */
|
|
9
|
+
export declare const LIST_FILE_NAMES: string[];
|
|
10
|
+
/**
|
|
11
|
+
* Directories never descended into when looking for memory content.
|
|
12
|
+
* This governs traversal only. A directory named here is still audited when the user
|
|
13
|
+
* points at it explicitly, which is how an archived corpus gets inspected.
|
|
14
|
+
*/
|
|
15
|
+
export declare const SKIP_DIRS: Set<string>;
|
|
16
|
+
export declare class DiscoveryError extends Error {
|
|
17
|
+
}
|
|
18
|
+
/** True for a regular file that is not a symlink. `lstat` on purpose: a cloned repo can ship a link. */
|
|
19
|
+
export declare function isPlainFile(abs: string): boolean;
|
|
20
|
+
/**
|
|
21
|
+
* Which tool surface a root can support, decided from directory entries alone.
|
|
22
|
+
*
|
|
23
|
+
* `createMcpServer` needs the shape of a root before it registers anything, on every launch.
|
|
24
|
+
* `discover()` cannot serve that: it reads the content of every memory file and folds in the
|
|
25
|
+
* auto-memory folder, which is most of the 310 to 380ms `doctor` cost the 2026-09-13 audit
|
|
26
|
+
* measured against a 68-file corpus. This probe does `stat` only, never a read, and tests the
|
|
27
|
+
* one distinction that changes the answer: `discover()` checks for `.minnimemory/` before it
|
|
28
|
+
* checks any other shape, so agreeing with it on a compiled root is the whole job. Every other
|
|
29
|
+
* shape it can return maps to the same "setup" surface.
|
|
30
|
+
*
|
|
31
|
+
* Every failure resolves to "setup": that surface holds the tools that diagnose and repair a
|
|
32
|
+
* root, so a root we cannot classify gets the tools that can tell the user why.
|
|
33
|
+
*
|
|
34
|
+
* Requires `manifest.json` inside `.minnimemory/` too (stat only, never read): a bare
|
|
35
|
+
* `.minnimemory/` directory with nothing compiled into it yet is `setup`, not `compiled`.
|
|
36
|
+
*
|
|
37
|
+
* `memory-dir` (added 2026-09-16, manifest-less recall): a root with no `.minnimemory/` at all
|
|
38
|
+
* that would still discover() to shape `memory-dir` or `auto-memory` - an index file (one of
|
|
39
|
+
* `LIST_FILE_NAMES`) plus topic files, three or more loose markdown files with no index, or a
|
|
40
|
+
* Claude Code auto-memory folder with content. Decided the same cheap way as the rest of this
|
|
41
|
+
* function: directory entries and a stat per candidate, never a file read, so this stays the
|
|
42
|
+
* probe discover() itself is too expensive to be. A root with a `.minnimemory/` directory (even
|
|
43
|
+
* an empty one, or one missing its manifest) never falls through to this check - `.minnimemory/`
|
|
44
|
+
* existing at all means "setup", exactly as before.
|
|
45
|
+
*/
|
|
46
|
+
export type WorkspacePhase = "compiled" | "memory-dir" | "setup";
|
|
47
|
+
export declare function detectPhase(root: string): WorkspacePhase;
|
|
48
|
+
export interface ImportResolution {
|
|
49
|
+
/** absolute paths of imported files inside the workspace root */
|
|
50
|
+
files: string[];
|
|
51
|
+
/**
|
|
52
|
+
* Imports that resolved outside the root and were not read. Reported as the raw text the
|
|
53
|
+
* memory file wrote, never as a resolved absolute path: the point is to not disclose where
|
|
54
|
+
* on this machine the target would have been.
|
|
55
|
+
*/
|
|
56
|
+
skipped: string[];
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Resolve the files a memory file pulls into the always-loaded prefix through `@path` imports,
|
|
60
|
+
* the way Claude Code does: relative to the importing file, `~/` allowed, recursive to a depth
|
|
61
|
+
* of five, each file once. A path that does not resolve is ignored rather than reported: an
|
|
62
|
+
* `@` in prose is not an error.
|
|
63
|
+
*
|
|
64
|
+
* Imports that land outside `root` are NOT followed unless `followExternal` is set.
|
|
65
|
+
*
|
|
66
|
+
* The memory file being audited is frequently one you did not write - the whole pitch is
|
|
67
|
+
* pointing this at a repo, including in CI - and an `@` line is content that repo controls.
|
|
68
|
+
* Following `@~/.claude/CLAUDE.md` from a cloned repo read seventeen files out of the auditor's
|
|
69
|
+
* home directory, whose headings, frontmatter, body-derived keywords and credential prefixes
|
|
70
|
+
* then surfaced in doctor and scan output. This is the same threat findHostFile already refuses
|
|
71
|
+
* symlinks for ("a cloned repo can point a memory file at anything on your machine"); imports
|
|
72
|
+
* were the second door to the same room.
|
|
73
|
+
*
|
|
74
|
+
* Following them remains correct for your own repo, where an out-of-root import genuinely is
|
|
75
|
+
* part of the prefix you pay for, so it stays available - as a decision the person running the
|
|
76
|
+
* tool makes, not one the scanned repo makes for them.
|
|
77
|
+
*/
|
|
78
|
+
export declare function resolveImports(hostAbs: string, root: string, followExternal?: boolean): ImportResolution;
|
|
79
|
+
/**
|
|
80
|
+
* A pointer stub is a host file whose entire content, headings and comments aside, names one
|
|
81
|
+
* other memory file: "@AGENTS.md", "AGENTS.md", "See `AGENTS.md` for the shared instructions".
|
|
82
|
+
* Eight of the twenty-three public CLAUDE.md files in the 2026-09-02 sweep were exactly this.
|
|
83
|
+
* Returns the absolute path of the named file when it exists next to the host, else undefined.
|
|
84
|
+
*/
|
|
85
|
+
export declare function pointerTargetOf(hostAbs: string, content: string): string | undefined;
|
|
86
|
+
/**
|
|
87
|
+
* Claude Code's own project-key algorithm, reverse-engineered empirically (not published), and
|
|
88
|
+
* verified against this machine's real `~/.claude/projects/` folder names across a dozen
|
|
89
|
+
* independent examples, including nested repos and a path containing `&`: every character that
|
|
90
|
+
* is not a-z, A-Z, or 0-9 becomes a single `-`. No collapsing of adjacent replacements, no case
|
|
91
|
+
* change. `C:\Code` -> `C--Code`; `C:\Code\Widgets\App` -> `C--Code-Widgets-App`.
|
|
92
|
+
*/
|
|
93
|
+
export declare function claudeProjectSlug(absPath: string): string;
|
|
94
|
+
/**
|
|
95
|
+
* Locate the auto-memory folder that belongs to `target`, the same way Claude Code itself would
|
|
96
|
+
* resolve it: git-repo-root if `target` is inside a repo, `target` itself otherwise, then
|
|
97
|
+
* Claude Code's own slug algorithm. Returns undefined, never throws, when nothing is found there -
|
|
98
|
+
* most targets simply have never been launched with Claude Code and that is not an error.
|
|
99
|
+
*/
|
|
100
|
+
export declare function locateAutoMemoryDir(target: string): string | undefined;
|
|
101
|
+
export interface DiscoverOptions {
|
|
102
|
+
/**
|
|
103
|
+
* Follow `@path` imports that resolve outside the workspace root. Off by default: the file
|
|
104
|
+
* being audited is often one you did not write. See resolveImports.
|
|
105
|
+
*/
|
|
106
|
+
followExternalImports?: boolean;
|
|
107
|
+
/**
|
|
108
|
+
* Fold in Claude Code's OS-level auto-memory folder (`~/.claude/projects/<slug>/memory/`),
|
|
109
|
+
* outside the target directory entirely. Default true - this is the two-location unification
|
|
110
|
+
* feature, and the CLI's own `doctor`/`bench` intentionally see it with no flag needed, since
|
|
111
|
+
* the operator running the CLI on their own machine and the memory it audits are the same
|
|
112
|
+
* party. Explicitly set false for the MCP server's scan/doctor tools: a target confined to
|
|
113
|
+
* "this server's root" per their own descriptions must not silently widen to the operator's
|
|
114
|
+
* entire global memory, the same category of reach `followExternalImports` already gates for
|
|
115
|
+
* `@path` imports (2026-09-10 audit).
|
|
116
|
+
*/
|
|
117
|
+
includeAutoMemory?: boolean;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Inspect `target` and build a Workspace.
|
|
121
|
+
*
|
|
122
|
+
* The important output is which files carry `alwaysLoaded: true`, because that set is what
|
|
123
|
+
* the whole tool is trying to shrink.
|
|
124
|
+
*/
|
|
125
|
+
export declare function discover(target: string, options?: DiscoverOptions): Workspace;
|