minnimemory 1.0.0-beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/LICENSE +39 -0
  2. package/README.md +824 -0
  3. package/dist/bench.d.ts +98 -0
  4. package/dist/bench.js +142 -0
  5. package/dist/benchReport.d.ts +12 -0
  6. package/dist/benchReport.js +128 -0
  7. package/dist/bounds.d.ts +40 -0
  8. package/dist/bounds.js +44 -0
  9. package/dist/cli.d.ts +15 -0
  10. package/dist/cli.js +503 -0
  11. package/dist/compile.d.ts +187 -0
  12. package/dist/compile.js +516 -0
  13. package/dist/discover.d.ts +125 -0
  14. package/dist/discover.js +520 -0
  15. package/dist/doctor.d.ts +9 -0
  16. package/dist/doctor.js +67 -0
  17. package/dist/episodic.d.ts +47 -0
  18. package/dist/episodic.js +130 -0
  19. package/dist/hook.d.ts +45 -0
  20. package/dist/hook.js +104 -0
  21. package/dist/index.d.ts +18 -0
  22. package/dist/index.js +18 -0
  23. package/dist/init.d.ts +125 -0
  24. package/dist/init.js +475 -0
  25. package/dist/instructions.d.ts +60 -0
  26. package/dist/instructions.js +270 -0
  27. package/dist/mcp.d.ts +109 -0
  28. package/dist/mcp.js +252 -0
  29. package/dist/mcpServer.d.ts +136 -0
  30. package/dist/mcpServer.js +997 -0
  31. package/dist/paths.d.ts +25 -0
  32. package/dist/paths.js +47 -0
  33. package/dist/recall.d.ts +113 -0
  34. package/dist/recall.js +256 -0
  35. package/dist/recallDir.d.ts +50 -0
  36. package/dist/recallDir.js +187 -0
  37. package/dist/reorganize.d.ts +62 -0
  38. package/dist/reorganize.js +216 -0
  39. package/dist/report.d.ts +16 -0
  40. package/dist/report.js +204 -0
  41. package/dist/router.d.ts +141 -0
  42. package/dist/router.js +314 -0
  43. package/dist/rules.d.ts +32 -0
  44. package/dist/rules.js +651 -0
  45. package/dist/scan.d.ts +110 -0
  46. package/dist/scan.js +173 -0
  47. package/dist/text.d.ts +158 -0
  48. package/dist/text.js +395 -0
  49. package/dist/tokenizer.d.ts +26 -0
  50. package/dist/tokenizer.js +69 -0
  51. package/dist/types.d.ts +156 -0
  52. package/dist/types.js +17 -0
  53. package/dist/version.d.ts +7 -0
  54. package/dist/version.js +7 -0
  55. package/dist/writeProtocol.d.ts +19 -0
  56. package/dist/writeProtocol.js +45 -0
  57. package/examples/CLAUDE.md +75 -0
  58. package/examples/README.md +7 -0
  59. package/package.json +52 -0
package/dist/text.js ADDED
@@ -0,0 +1,395 @@
1
+ /**
2
+ * Text primitives shared by the rule engine and the compiler.
3
+ */
4
+ // No lazy capture followed by `\s*$`: that pair is quadratic on a heading padded with
5
+ // thousands of spaces (security audit 2026-09-02). Trim in code instead.
6
+ const RE_HEADING = /^(#{1,6})\s+(.*)$/;
7
+ const RE_FENCE = /^\s*(?:```|~~~)/;
8
+ /**
9
+ * Split lines into heading-delimited sections. Lines inside fenced code blocks are never
10
+ * headings: a shell comment like `# Machine-readable output` in a bash example is not a title.
11
+ * Found by the 2026-09-02 sweep, where one real CLAUDE.md gained a phantom H1 mid-file and
12
+ * half its content was duplicated into AlwaysOnMemory as "preamble".
13
+ */
14
+ export function sections(lines) {
15
+ const out = [];
16
+ let current;
17
+ let inFence = false;
18
+ lines.forEach((line, i) => {
19
+ if (RE_FENCE.test(line)) {
20
+ inFence = !inFence;
21
+ return;
22
+ }
23
+ if (inFence)
24
+ return;
25
+ const m = RE_HEADING.exec(line);
26
+ if (!m)
27
+ return;
28
+ if (current) {
29
+ current.endLine = i;
30
+ out.push(current);
31
+ }
32
+ current = {
33
+ heading: (m[2] ?? "").trim(),
34
+ level: (m[1] ?? "#").length,
35
+ startLine: i + 1,
36
+ endLine: lines.length,
37
+ };
38
+ });
39
+ if (current)
40
+ out.push(current);
41
+ if (out.length === 0) {
42
+ out.push({ heading: "(document)", level: 0, startLine: 1, endLine: lines.length });
43
+ }
44
+ return out;
45
+ }
46
+ /** Split lines into blank-line-delimited blocks. */
47
+ export function blocks(lines) {
48
+ const out = [];
49
+ let buf = [];
50
+ let start = 1;
51
+ const flush = (endLine) => {
52
+ const text = buf.join("\n").trim();
53
+ if (text)
54
+ out.push({ text, startLine: start, endLine });
55
+ buf = [];
56
+ };
57
+ lines.forEach((line, i) => {
58
+ if (line.trim() === "") {
59
+ flush(i);
60
+ start = i + 2;
61
+ }
62
+ else {
63
+ if (buf.length === 0)
64
+ start = i + 1;
65
+ buf.push(line);
66
+ }
67
+ });
68
+ flush(lines.length);
69
+ return out;
70
+ }
71
+ /** Normalise a block so cosmetic differences do not hide a duplicate. */
72
+ export function normalise(text) {
73
+ return text
74
+ .split("\n")
75
+ .map((l) => l.replace(/^\s*(?:[-*+]|\d+\.|#{1,6})\s*/, "").trim())
76
+ .join(" ")
77
+ .replace(/\s+/g, " ")
78
+ .toLowerCase()
79
+ .trim();
80
+ }
81
+ /** Filesystem-safe, stable slug for a heading. */
82
+ export function slugify(heading) {
83
+ const base = heading
84
+ .toLowerCase()
85
+ .replace(/[`*_~]/g, "")
86
+ .replace(/[^a-z0-9]+/g, "_")
87
+ .replace(/^_+|_+$/g, "")
88
+ .replace(/_{2,}/g, "_");
89
+ const capped = base.length > 40 ? base.slice(0, 40).replace(/_[^_]*$/, "") : base;
90
+ return capped || base || "section";
91
+ }
92
+ /**
93
+ * Volatility: does this content change often, so that editing it would invalidate the cached
94
+ * always-loaded prefix?
95
+ *
96
+ * This is scored on evidence, not fired by a single word. The 2026-09-02 sweep over 23 real
97
+ * CLAUDE.md/AGENTS.md files from public repos showed that one-word matching flagged stable
98
+ * sections constantly: a version number inside a URL, "the next step" in a coding standard,
99
+ * "regenerate the changelog" in commit guidance, "pending activity" in a debugging note. Those
100
+ * are prose, not status. What actually marks status is a status heading, dated log entries,
101
+ * checklists, or several status phrases together.
102
+ *
103
+ * Scoring (a section is volatile at VOLATILITY_THRESHOLD or above):
104
+ * status heading ("Current status", "Changelog", "Roadmap", "Open items"...) +2
105
+ * list item that starts with a date (a log entry) +1 each
106
+ * strong status phrase ("as of 2026", "in progress", "blocked on") +1 each
107
+ * date anywhere else in a line +0.5 each
108
+ * weak status word ("currently", "pending", "todo", "roadmap" in prose) +0.5 each
109
+ * checkbox item +0.25 each, at most +1
110
+ * version string +0.25 each
111
+ * Inline code spans and URLs are removed from a line before matching, so a version in a link
112
+ * or a `TODO` in a code sample never counts.
113
+ */
114
+ export const VOLATILITY_THRESHOLD = 2;
115
+ /**
116
+ * Interface rule O7, the granularity clause: a section above this many tokens carries
117
+ * subheadings so that retrieval has a unit smaller than the whole. `doctor` flags a flat
118
+ * section over it (MM009); `init` splits a section over it at its subheadings into separate
119
+ * OnDemandMemory files, so the agent-driven route (an agent reading a whole file the
120
+ * OnDemandMemory list named) gets the same granularity `recall` gets from in-file sections.
121
+ */
122
+ export const GRANULARITY_TOKENS = 800;
123
+ /**
124
+ * A bullet (list marker or numbered item) whose first few words carry a date: a changelog or
125
+ * session-log entry. The single shared definition (2026-09-13 audit round 2, D7a) - episodic.ts,
126
+ * router.ts and rules.ts's MM009 each grew their own slightly different copy (one required a full
127
+ * day, one skipped numbered lists, one had an explicit checkbox branch this pattern's lazy
128
+ * 0-24-character window already covers). This is the widest of the three: numbered lists
129
+ * (`1.`/`1)`), a month-only date, and up to 24 characters (a checkbox, a bold marker, ...) before
130
+ * the date all count.
131
+ */
132
+ export const RE_DATED_BULLET = /^\s*(?:[-*+]|\d+[.)])\s+(?:\*\*)?[^\n]{0,24}?\b20\d{2}-\d{2}(?:-\d{2})?\b/;
133
+ const RE_ANY_BULLET = /^\s*(?:[-*+]|\d+[.)])\s+/;
134
+ /**
135
+ * Is this body of lines "episodic": mostly a list of dated entries (a changelog, a session log)?
136
+ * At least 3 dated bullets, and at least half of all bullets dated. The one rule every episodic-
137
+ * detection call site in the codebase now shares, so "how many dated entries make a section a
138
+ * changelog" has one answer instead of three.
139
+ */
140
+ export function isEpisodicBody(lines) {
141
+ const bullets = lines.filter((l) => RE_ANY_BULLET.test(l));
142
+ const dated = bullets.filter((l) => RE_DATED_BULLET.test(l));
143
+ return dated.length >= 3 && dated.length * 2 >= bullets.length;
144
+ }
145
+ // "status" counts only when it is the topic of the heading ("Status", "Current status",
146
+ // "Project status"), not a word inside one ("PR Status (CI Failures and Reviews)").
147
+ const RE_STATUS_HEADING = /(?:^\s*#*\s*(?:[\w-]+\s){0,2}status\s*$)|\b(?:changelog|change log|roadmap|to-?dos?|open items?|next steps|known issues|in progress|wip|backlog|progress log|session log|current (?:state|work|focus|sprint|plan))\b/i;
148
+ const RE_DATE = /\b20\d{2}-\d{2}-\d{2}\b/g;
149
+ const RE_CHECKBOX = /^\s*[-*+]\s*\[[ xX]\]/;
150
+ const RE_STATUS_STRONG = /\b(?:as of (?:20\d{2}|today|now|this)|in progress|current (?:status|state|sprint|plan)|todo:|wip:|blocked on|still (?:open|pending|unresolved|blocked))\b/gi;
151
+ // "currently" is weak: "currently unsupported" is a stable fact far more often than a status.
152
+ const RE_STATUS_WEAK = /\b(?:currently|pending|next steps|changelog|roadmap|open items?|todo|wip)\b/gi;
153
+ /** Checkboxes alone never decide: a PR-review checklist template is stable content. */
154
+ const CHECKLIST_CAP = 1;
155
+ const RE_VERSION = /\bv\d+\.\d+\.\d+\b/g;
156
+ const RE_INLINE_CODE = /`[^`\n]*`/g;
157
+ const RE_URL = /\bhttps?:\/\/\S+/g;
158
+ function countMatches(re, text) {
159
+ re.lastIndex = 0;
160
+ let n = 0;
161
+ while (re.exec(text))
162
+ n++;
163
+ return n;
164
+ }
165
+ /**
166
+ * Score a run of lines. Pass the heading line as `heading` when the lines are a section body;
167
+ * pass an empty string for a headingless block. A heading with no body is never volatile: a
168
+ * title cannot be edited often when there is nothing under it to edit.
169
+ */
170
+ export function volatility(heading, bodyLines) {
171
+ const body = bodyLines.join("\n");
172
+ if (body.trim() === "")
173
+ return { score: 0, reasons: [] };
174
+ let score = 0;
175
+ const found = { heading: 0, dated: 0, checklist: 0, status: 0, version: 0 };
176
+ if (RE_STATUS_HEADING.test(heading)) {
177
+ score += 2;
178
+ found.heading++;
179
+ }
180
+ for (const raw of bodyLines) {
181
+ const line = raw.replace(RE_INLINE_CODE, " ").replace(RE_URL, " ");
182
+ if (line.trim() === "")
183
+ continue;
184
+ if (RE_DATED_BULLET.test(line)) {
185
+ score += 1;
186
+ found.dated++;
187
+ }
188
+ else {
189
+ const dates = countMatches(RE_DATE, line);
190
+ if (dates > 0) {
191
+ score += 0.5 * dates;
192
+ found.dated += dates;
193
+ }
194
+ }
195
+ if (RE_CHECKBOX.test(line)) {
196
+ if (found.checklist * 0.25 < CHECKLIST_CAP)
197
+ score += 0.25;
198
+ found.checklist++;
199
+ }
200
+ const strong = countMatches(RE_STATUS_STRONG, line);
201
+ if (strong > 0) {
202
+ score += strong;
203
+ found.status += strong;
204
+ }
205
+ const weak = countMatches(RE_STATUS_WEAK, line);
206
+ if (weak > 0) {
207
+ score += 0.5 * weak;
208
+ found.status += weak;
209
+ }
210
+ const versions = countMatches(RE_VERSION, line);
211
+ if (versions > 0) {
212
+ score += 0.25 * versions;
213
+ found.version += versions;
214
+ }
215
+ }
216
+ const reasons = [];
217
+ if (found.heading)
218
+ reasons.push("a status heading");
219
+ if (found.dated)
220
+ reasons.push("dated entries");
221
+ if (found.checklist)
222
+ reasons.push("checklist items");
223
+ if (found.status)
224
+ reasons.push("status language");
225
+ if (found.version)
226
+ reasons.push("a version string");
227
+ return { score, reasons };
228
+ }
229
+ /** Reasons a headingless block is volatile; empty when it is not. */
230
+ export function volatilityReasons(text) {
231
+ const v = volatility("", text.split("\n"));
232
+ return v.score >= VOLATILITY_THRESHOLD ? v.reasons : [];
233
+ }
234
+ /** Reasons a heading-led section is volatile; empty when it is not. */
235
+ export function sectionVolatilityReasons(sectionLines) {
236
+ const v = volatility(sectionLines[0] ?? "", sectionLines.slice(1));
237
+ return v.score >= VOLATILITY_THRESHOLD ? v.reasons : [];
238
+ }
239
+ /** Delimiters around the generated regions of a stub: the OnDemandMemory list and the instruction block. */
240
+ export const LIST_START = "<!-- minnimemory:ondemand-list -->";
241
+ export const LIST_END = "<!-- /minnimemory:ondemand-list -->";
242
+ export const INSTRUCTIONS_START = "<!-- minnimemory:instructions -->";
243
+ export const INSTRUCTIONS_END = "<!-- /minnimemory:instructions -->";
244
+ /**
245
+ * Locate one generated region. Only a single well-formed pair counts: the first opening marker
246
+ * that is followed by a closing marker. An unclosed opener, a second pair, or a closer with no
247
+ * opener is ignored, so a marker pasted into prose cannot hide the rest of a file from the
248
+ * rules (security audit 2026-09-02, finding 5). Returns 0-based inclusive line indices.
249
+ */
250
+ export function generatedRegion(lines, start, end) {
251
+ const from = lines.findIndex((l) => l.trim() === start);
252
+ if (from < 0)
253
+ return undefined;
254
+ const to = lines.findIndex((l, i) => i > from && l.trim() === end);
255
+ if (to < 0)
256
+ return undefined;
257
+ return { from, to };
258
+ }
259
+ /**
260
+ * Blank out the generated regions, line for line so line numbers are preserved. The
261
+ * OnDemandMemory list is a routing manifest: its trigger words ("changelog", "open items") are
262
+ * labels for OnDemandMemory files, not status, and it only changes when a file is added or
263
+ * removed. The instruction block is fixed text. Auditing either as prose would make `doctor`
264
+ * flag `init`'s own output.
265
+ */
266
+ export function stripGeneratedRegions(lines) {
267
+ const out = [...lines];
268
+ for (const [s, e] of [
269
+ [LIST_START, LIST_END],
270
+ [INSTRUCTIONS_START, INSTRUCTIONS_END],
271
+ ]) {
272
+ const r = generatedRegion(out, s, e);
273
+ if (!r)
274
+ continue;
275
+ for (let i = r.from; i <= r.to; i++)
276
+ out[i] = "";
277
+ }
278
+ return out;
279
+ }
280
+ /** Kept for callers that only know about the OnDemandMemory list region. */
281
+ export const stripGeneratedList = stripGeneratedRegions;
282
+ /**
283
+ * The inverse of `stripGeneratedRegions`: blank every line outside one marked region, keeping
284
+ * the region's own lines verbatim at their real position. Line numbers stay meaningful for a
285
+ * region carved out of a larger always-loaded file (compile.ts's merged AlwaysOnMemory.md holds
286
+ * always-on prose and the OnDemandMemory list in one file; this is how a caller gets just the
287
+ * list's lines back without losing the line numbers a finding reports against the real file).
288
+ */
289
+ export function regionOnly(lines, start, end) {
290
+ const region = generatedRegion(lines, start, end);
291
+ if (!region)
292
+ return lines.map(() => "");
293
+ return lines.map((l, i) => (i >= region.from && i <= region.to ? l : ""));
294
+ }
295
+ /**
296
+ * Leading YAML frontmatter: a `---` line first, closed by the next `---` line. Returned as the
297
+ * verbatim lines plus the 1-indexed line number where the document body begins. Frontmatter has
298
+ * to stay first in a file to remain frontmatter, so the compiler carries it through untouched
299
+ * instead of treating it as preamble prose.
300
+ */
301
+ export function splitFrontmatter(lines) {
302
+ if ((lines[0] ?? "").trim() !== "---")
303
+ return { frontmatter: [], bodyStartLine: 1 };
304
+ for (let i = 1; i < lines.length; i++) {
305
+ if ((lines[i] ?? "").trim() === "---") {
306
+ return { frontmatter: lines.slice(0, i + 1), bodyStartLine: i + 2 };
307
+ }
308
+ }
309
+ // An opening fence with no close is not frontmatter, just a line of dashes.
310
+ return { frontmatter: [], bodyStartLine: 1 };
311
+ }
312
+ export const STOPWORDS = new Set([
313
+ "the", "and", "for", "that", "this", "with", "from", "are", "was", "were", "you", "your", "our", "its", "it",
314
+ "not", "but", "all", "any", "can", "has", "have", "had", "will", "would", "should", "must", "may", "when",
315
+ "what", "which", "who", "how", "why", "into", "over", "under", "then", "than", "them", "they", "their",
316
+ "itself", "himself", "herself", "themselves", "yourself", "yourselves", "ourselves",
317
+ "there", "here", "only", "also", "just", "use", "used", "using", "via", "per", "one", "two", "three", "new",
318
+ "run", "see", "get", "set", "add", "make", "made", "does", "did", "done", "been", "being", "each", "every",
319
+ "some", "such", "same", "other", "more", "most", "less", "least", "very", "much", "many", "few", "own",
320
+ "about", "after", "before", "between", "during", "without", "within", "because", "while", "where",
321
+ "these", "those", "both", "either", "neither", "however", "therefore", "instead", "rather", "always",
322
+ "never", "often", "already", "still", "yet", "once", "first", "last", "next", "file", "files", "line",
323
+ "lines", "code", "project", "repo", "repository", "note", "notes", "md", "http", "https", "www", "com",
324
+ // contraction fragments left behind by word splitting
325
+ "don", "doesn", "isn", "aren", "wasn", "weren", "won", "wouldn", "couldn", "shouldn", "didn", "hasn",
326
+ "haven", "hadn", "ll", "ve", "re", "ain", "let", "its",
327
+ // generic filler that ranks high on frequency but routes nowhere
328
+ "actually", "enough", "happens", "result", "results", "content", "thing", "things", "way", "ways",
329
+ "need", "needs", "want", "wants", "give", "given", "gets", "goes", "going", "take", "takes", "comes",
330
+ "asked", "asks", "answer", "answers", "said", "says", "tell", "tells", "show", "shows", "know", "known",
331
+ "something", "anything", "everything", "nothing", "someone", "anyone", "everyone", "stuff", "kind",
332
+ "sort", "lot", "bit", "good", "bad", "best", "worse", "worst", "real", "really", "quite", "pretty",
333
+ // seen ranking as triggers in the 2026-09-02 real-world sweep; none of them route anywhere
334
+ "etc", "foo", "bar", "baz", "important", "key", "existing", "adding", "acting", "case", "empty",
335
+ "closest", "check", "change", "changes", "changed", "following", "above", "below", "example",
336
+ "examples", "section", "sections", "item", "items", "step", "steps", "open", "closed", "list",
337
+ // the connective in a merged OnDemandMemory file's heading ("Common tasks through Changelog")
338
+ "through",
339
+ ]);
340
+ /** A code token's raw frequency inside a fenced block, relative to a comment or prose word. */
341
+ export const CODE_WEIGHT = 0.2;
342
+ /**
343
+ * Split body text into (fragment, weight) pieces for keyword extraction: prose and `#` comment
344
+ * text keep weight 1, the command/flag/tool-name text around a comment inside a fenced code
345
+ * block is down-weighted, so a repeated tool name doesn't drown out a rare-but-specific comment
346
+ * word under raw frequency. Found live (2026-09-09): an OnDemandMemory file whose only mentions of "UAC" and
347
+ * "window" were in comments lost to "npm"/"venv"/"pester" appearing many times across unrelated
348
+ * commands sharing one section, so a question phrased the way a user actually asks it ("run
349
+ * without a UAC prompt") matched no trigger at all. Shared by `keywords()` below and
350
+ * `router.ts`'s `selectTriggers()`, the two places that rank candidate routing words this way.
351
+ */
352
+ export function weightedFragments(body) {
353
+ const out = [];
354
+ let inFence = false;
355
+ for (const line of body.split(/\r?\n/)) {
356
+ if (RE_FENCE.test(line)) {
357
+ inFence = !inFence;
358
+ continue;
359
+ }
360
+ if (!inFence) {
361
+ out.push({ text: line, weight: 1 });
362
+ continue;
363
+ }
364
+ const hash = line.indexOf("#");
365
+ if (hash === -1) {
366
+ out.push({ text: line, weight: CODE_WEIGHT });
367
+ continue;
368
+ }
369
+ out.push({ text: line.slice(0, hash), weight: CODE_WEIGHT });
370
+ out.push({ text: line.slice(hash + 1), weight: 1 });
371
+ }
372
+ return out;
373
+ }
374
+ /**
375
+ * Pull candidate routing keywords out of a piece of text.
376
+ * Frequency-ranked, stopword-filtered, and biased toward terms in the heading.
377
+ */
378
+ export function keywords(heading, body, limit = 6) {
379
+ const counts = new Map();
380
+ const add = (source, weight) => {
381
+ const words = source.toLowerCase().match(/[a-z][a-z0-9_-]{2,}/g) ?? [];
382
+ for (const w of words) {
383
+ if (STOPWORDS.has(w))
384
+ continue;
385
+ counts.set(w, (counts.get(w) ?? 0) + weight);
386
+ }
387
+ };
388
+ add(heading, 5);
389
+ for (const { text, weight } of weightedFragments(body))
390
+ add(text, weight);
391
+ return [...counts.entries()]
392
+ .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))
393
+ .slice(0, limit)
394
+ .map(([w]) => w);
395
+ }
@@ -0,0 +1,26 @@
1
+ /**
2
+ * Offline token estimation.
3
+ *
4
+ * `doctor` and `init` must work with no network access and no API key, so the default
5
+ * tokenizer is a heuristic rather than a real BPE encoder.
6
+ *
7
+ * approx-v2, calibrated 2026-09-02. The rules below were fitted against OpenAI's o200k_base
8
+ * encoder (the cl100k_base result is within 1%) over 73 real memory files: 23 CLAUDE.md and
9
+ * AGENTS.md files from public repositories plus a 50-file personal Claude Code memory
10
+ * directory, 240 KB in total. Result on whole files: estimate / o200k mean 1.00, standard
11
+ * deviation 0.04, range 0.91 to 1.08. The previous approx-v1 overstated by 43% on the same
12
+ * corpus, worst on code-heavy files, because it charged every punctuation character a whole
13
+ * token and every four letters a token.
14
+ *
15
+ * Anthropic does not publish the tokenizer for current Claude models, so this is calibrated
16
+ * to a proxy, and figures the tool prints are comparable with each other rather than exact
17
+ * for any one model. Numbers are always labelled with this tokenizer id.
18
+ */
19
+ export declare const TOKENIZER_ID = "approx-v2";
20
+ /**
21
+ * Estimate the number of tokens in a string.
22
+ * Deterministic, allocation-light, and safe on empty input.
23
+ */
24
+ export declare function estimateTokens(text: string): number;
25
+ /** Format a token count for display, for example 11240 -> "11,240". */
26
+ export declare function formatTokens(n: number): string;
@@ -0,0 +1,69 @@
1
+ /**
2
+ * Offline token estimation.
3
+ *
4
+ * `doctor` and `init` must work with no network access and no API key, so the default
5
+ * tokenizer is a heuristic rather than a real BPE encoder.
6
+ *
7
+ * approx-v2, calibrated 2026-09-02. The rules below were fitted against OpenAI's o200k_base
8
+ * encoder (the cl100k_base result is within 1%) over 73 real memory files: 23 CLAUDE.md and
9
+ * AGENTS.md files from public repositories plus a 50-file personal Claude Code memory
10
+ * directory, 240 KB in total. Result on whole files: estimate / o200k mean 1.00, standard
11
+ * deviation 0.04, range 0.91 to 1.08. The previous approx-v1 overstated by 43% on the same
12
+ * corpus, worst on code-heavy files, because it charged every punctuation character a whole
13
+ * token and every four letters a token.
14
+ *
15
+ * Anthropic does not publish the tokenizer for current Claude models, so this is calibrated
16
+ * to a proxy, and figures the tool prints are comparable with each other rather than exact
17
+ * for any one model. Numbers are always labelled with this tokenizer id.
18
+ */
19
+ export const TOKENIZER_ID = "approx-v2";
20
+ const RE_PIECES = /\s+|[A-Za-z]+|[0-9]+|[^\sA-Za-z0-9]/g;
21
+ /** Fitted so the corpus mean lands on 1.00; the rule set alone lands on 0.95. */
22
+ const CALIBRATION = 1.05;
23
+ /**
24
+ * Estimate the number of tokens in a string.
25
+ * Deterministic, allocation-light, and safe on empty input.
26
+ */
27
+ export function estimateTokens(text) {
28
+ if (!text)
29
+ return 0;
30
+ const pieces = text.match(RE_PIECES);
31
+ if (!pieces)
32
+ return 0;
33
+ let tokens = 0;
34
+ let previousSymbol = null;
35
+ for (const piece of pieces) {
36
+ const first = piece.charCodeAt(0);
37
+ // Whitespace run: a single space merges into the following word; each newline costs one.
38
+ if (piece.trimStart() === "") {
39
+ for (let i = 0; i < piece.length; i++) {
40
+ if (piece.charCodeAt(i) === 10)
41
+ tokens++;
42
+ }
43
+ previousSymbol = null;
44
+ continue;
45
+ }
46
+ // Letters: common words up to about seven letters are one token; longer words split
47
+ // roughly every five letters after that.
48
+ if ((first >= 65 && first <= 90) || (first >= 97 && first <= 122)) {
49
+ tokens += 1 + Math.floor(Math.max(0, piece.length - 7) / 5);
50
+ previousSymbol = null;
51
+ continue;
52
+ }
53
+ // Digits: encoders group numbers in runs of up to three digits.
54
+ if (first >= 48 && first <= 57) {
55
+ tokens += Math.ceil(piece.length / 3);
56
+ previousSymbol = null;
57
+ continue;
58
+ }
59
+ // A punctuation or symbol character. Most merge with a neighbour (":/" "()" "**"), so one
60
+ // costs well under a token, and a repeat of the same symbol ("---", "```") costs less again.
61
+ tokens += previousSymbol === piece ? 0.35 : 0.7;
62
+ previousSymbol = piece;
63
+ }
64
+ return Math.round(tokens * CALIBRATION);
65
+ }
66
+ /** Format a token count for display, for example 11240 -> "11,240". */
67
+ export function formatTokens(n) {
68
+ return n.toLocaleString("en-US");
69
+ }
@@ -0,0 +1,156 @@
1
+ /**
2
+ * Shared types for the minnimemory rule engine.
3
+ */
4
+ export type Severity = "low" | "med" | "high";
5
+ export declare const SEVERITY_ORDER: Record<Severity, number>;
6
+ /** What role a memory file plays in the context window. */
7
+ export type FileKind =
8
+ /** the host's own memory file, for example CLAUDE.md */
9
+ "host"
10
+ /** compiled AlwaysOnMemory body */
11
+ | "alwaysOn"
12
+ /** compiled OnDemandMemory list */
13
+ | "onDemandList"
14
+ /** compiled OnDemandMemory file */
15
+ | "onDemand"
16
+ /** an uncompiled memory file sitting loose in a directory */
17
+ | "loose"
18
+ /** a file the host file pulls in: a Claude Code `@path` import, or the target of a pointer stub */
19
+ | "import";
20
+ export interface MemoryFile {
21
+ abs: string;
22
+ /** path relative to the workspace root, always with forward slashes */
23
+ rel: string;
24
+ content: string;
25
+ lines: string[];
26
+ tokens: number;
27
+ /** true when this file is re-sent to the model on every turn */
28
+ alwaysLoaded: boolean;
29
+ kind: FileKind;
30
+ /**
31
+ * True when this file's content is intentionally mirrored into a generated stub.
32
+ * Duplication rules skip these: the copy is the design, not a defect.
33
+ */
34
+ generated?: boolean;
35
+ }
36
+ export type WorkspaceShape =
37
+ /** a single memory file passed directly */
38
+ "single-file"
39
+ /** a repo with a host memory file at its root */
40
+ | "host-file"
41
+ /** a repo already compiled into .minnimemory/ */
42
+ | "compiled"
43
+ /** a directory of memory files, with or without an OnDemandMemory list */
44
+ | "memory-dir"
45
+ /** a project with no host file of its own whose Claude Code auto-memory folder exists */
46
+ | "auto-memory";
47
+ export interface Workspace {
48
+ root: string;
49
+ shape: WorkspaceShape;
50
+ files: MemoryFile[];
51
+ /** the OnDemandMemory list, when one was found */
52
+ onDemandList?: MemoryFile;
53
+ /**
54
+ * The auto-memory folder's own MEMORY.md, when that folder was folded in. It governs the
55
+ * `(auto-memory)/` topic files, while `onDemandList` governs compiled OnDemandMemory files;
56
+ * MM004 and MM006 check each file against the list that actually routes to it.
57
+ */
58
+ autoMemoryList?: MemoryFile;
59
+ /**
60
+ * The auto-memory folder folded into this workspace, when Claude Code's own project-key
61
+ * algorithm resolves one for `root` (or its git repo root) and it exists on disk. A genuinely
62
+ * different directory than `root` - see discover.ts's locateAutoMemoryDir(). Undefined means
63
+ * none was found, not that the lookup was skipped.
64
+ */
65
+ autoMemoryRoot?: string;
66
+ /**
67
+ * Set when the host file is a pointer stub: a tiny file whose whole content names another
68
+ * memory file ("@AGENTS.md", "AGENTS.md", "See `AGENTS.md`"). The named file is what the
69
+ * agent actually reads, so it is included as an always-loaded import and `init` compiles it
70
+ * instead of the stub. Workspace-relative path of that target.
71
+ */
72
+ pointerTarget?: string;
73
+ /**
74
+ * The compiled workspace's manifest, parsed, when the shape is `compiled` and the file is
75
+ * valid JSON. Rules that compare what is on disk against what `init` wrote (MM010, drift)
76
+ * read it from here; nothing else does, and an unreadable manifest simply leaves it unset.
77
+ */
78
+ manifest?: CompiledManifest;
79
+ /**
80
+ * `@path` imports that pointed outside the root and were not followed. Present so a report
81
+ * can say the always-loaded total is incomplete rather than quietly under-counting it.
82
+ */
83
+ skippedImports?: string[];
84
+ }
85
+ /** The subset of manifest.json the rule engine reads; the full shape lives in compile.ts. */
86
+ export interface CompiledManifest {
87
+ sourceHash: string;
88
+ always: {
89
+ file: string;
90
+ hash: string;
91
+ };
92
+ /** absent in a workspace compiled before manifest v3, so MM010 checks it only when present */
93
+ stub?: {
94
+ file: string;
95
+ hash: string;
96
+ };
97
+ onDemandFiles: {
98
+ name: string;
99
+ file: string;
100
+ hash: string;
101
+ }[];
102
+ }
103
+ export interface Finding {
104
+ rule: string;
105
+ severity: Severity;
106
+ message: string;
107
+ /** workspace-relative path */
108
+ file: string;
109
+ /** 1-indexed, inclusive */
110
+ startLine?: number;
111
+ endLine?: number;
112
+ /** tokens implicated by this finding, when the number is meaningful */
113
+ tokens?: number;
114
+ /** short suggested remedy */
115
+ hint?: string;
116
+ /** true for the synthetic "and N more" marker, which always sorts last in its group */
117
+ overflow?: boolean;
118
+ }
119
+ export interface DoctorConfig {
120
+ /** token ceiling for the always-loaded prefix (MM001) */
121
+ budget: number;
122
+ /** rule ids to skip */
123
+ ignore: string[];
124
+ /** when non-empty, run only these rule ids */
125
+ only: string[];
126
+ /** exit non-zero at this severity or above */
127
+ failOn: Severity;
128
+ /** OnDemandMemory list line character cap (MM006) */
129
+ maxListLine: number;
130
+ /** follow `@path` imports that resolve outside the workspace root (off by default) */
131
+ followExternalImports: boolean;
132
+ /** fold in the OS-level Claude Code auto-memory folder (on by default - see DiscoverOptions) */
133
+ includeAutoMemory: boolean;
134
+ }
135
+ export declare const DEFAULT_CONFIG: DoctorConfig;
136
+ export interface RuleContext {
137
+ workspace: Workspace;
138
+ config: DoctorConfig;
139
+ }
140
+ export interface Rule {
141
+ id: string;
142
+ severity: Severity;
143
+ title: string;
144
+ run(ctx: RuleContext): Finding[];
145
+ }
146
+ export interface DoctorResult {
147
+ workspace: Workspace;
148
+ findings: Finding[];
149
+ /** total tokens re-sent on every turn */
150
+ alwaysLoadedTokens: number;
151
+ /** number of files contributing to the always-loaded prefix */
152
+ alwaysLoadedFiles: number;
153
+ tokenizer: string;
154
+ }
155
+ /** Files that are re-sent on every turn. */
156
+ export declare function alwaysLoaded(ws: Workspace): MemoryFile[];
package/dist/types.js ADDED
@@ -0,0 +1,17 @@
1
+ /**
2
+ * Shared types for the minnimemory rule engine.
3
+ */
4
+ export const SEVERITY_ORDER = { low: 0, med: 1, high: 2 };
5
+ export const DEFAULT_CONFIG = {
6
+ budget: 2000,
7
+ ignore: [],
8
+ only: [],
9
+ failOn: "high",
10
+ maxListLine: 120,
11
+ followExternalImports: false,
12
+ includeAutoMemory: true,
13
+ };
14
+ /** Files that are re-sent on every turn. */
15
+ export function alwaysLoaded(ws) {
16
+ return ws.files.filter((f) => f.alwaysLoaded);
17
+ }
@@ -0,0 +1,7 @@
1
+ /**
2
+ * The single source of the version string. Before this file, "0.1.0" was hand-typed in four
3
+ * places (cli.ts's own VERSION constant, and both McpServer constructors in mcpServer.ts), which
4
+ * is how they drift: a bump to one place with the others forgotten is silent, since nothing
5
+ * compares them.
6
+ */
7
+ export declare const VERSION = "1.0.0-beta.1";