minnimemory 1.0.0-beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/LICENSE +39 -0
  2. package/README.md +824 -0
  3. package/dist/bench.d.ts +98 -0
  4. package/dist/bench.js +142 -0
  5. package/dist/benchReport.d.ts +12 -0
  6. package/dist/benchReport.js +128 -0
  7. package/dist/bounds.d.ts +40 -0
  8. package/dist/bounds.js +44 -0
  9. package/dist/cli.d.ts +15 -0
  10. package/dist/cli.js +503 -0
  11. package/dist/compile.d.ts +187 -0
  12. package/dist/compile.js +516 -0
  13. package/dist/discover.d.ts +125 -0
  14. package/dist/discover.js +520 -0
  15. package/dist/doctor.d.ts +9 -0
  16. package/dist/doctor.js +67 -0
  17. package/dist/episodic.d.ts +47 -0
  18. package/dist/episodic.js +130 -0
  19. package/dist/hook.d.ts +45 -0
  20. package/dist/hook.js +104 -0
  21. package/dist/index.d.ts +18 -0
  22. package/dist/index.js +18 -0
  23. package/dist/init.d.ts +125 -0
  24. package/dist/init.js +475 -0
  25. package/dist/instructions.d.ts +60 -0
  26. package/dist/instructions.js +270 -0
  27. package/dist/mcp.d.ts +109 -0
  28. package/dist/mcp.js +252 -0
  29. package/dist/mcpServer.d.ts +136 -0
  30. package/dist/mcpServer.js +997 -0
  31. package/dist/paths.d.ts +25 -0
  32. package/dist/paths.js +47 -0
  33. package/dist/recall.d.ts +113 -0
  34. package/dist/recall.js +256 -0
  35. package/dist/recallDir.d.ts +50 -0
  36. package/dist/recallDir.js +187 -0
  37. package/dist/reorganize.d.ts +62 -0
  38. package/dist/reorganize.js +216 -0
  39. package/dist/report.d.ts +16 -0
  40. package/dist/report.js +204 -0
  41. package/dist/router.d.ts +141 -0
  42. package/dist/router.js +314 -0
  43. package/dist/rules.d.ts +32 -0
  44. package/dist/rules.js +651 -0
  45. package/dist/scan.d.ts +110 -0
  46. package/dist/scan.js +173 -0
  47. package/dist/text.d.ts +158 -0
  48. package/dist/text.js +395 -0
  49. package/dist/tokenizer.d.ts +26 -0
  50. package/dist/tokenizer.js +69 -0
  51. package/dist/types.d.ts +156 -0
  52. package/dist/types.js +17 -0
  53. package/dist/version.d.ts +7 -0
  54. package/dist/version.js +7 -0
  55. package/dist/writeProtocol.d.ts +19 -0
  56. package/dist/writeProtocol.js +45 -0
  57. package/examples/CLAUDE.md +75 -0
  58. package/examples/README.md +7 -0
  59. package/package.json +52 -0
@@ -0,0 +1,141 @@
1
+ /**
2
+ * The router: deterministic lexical ranking over section-level units (interface rule O4).
3
+ *
4
+ * Two decisions here, both from the 2026-09-04 interface audit
5
+ * (interface audit, 2026-09-04):
6
+ *
7
+ * 1. The retrieval unit is a section, not an OnDemandMemory file. A file-level hit on a real
8
+ * memory file cost 1,303 to 8,825 tokens, and the largest file was the changelog, so better
9
+ * recall at file granularity opened exactly the content whose non-opening produced the
10
+ * saving. At section granularity the same queries cost 17 to 60 tokens where the file had
11
+ * structure. Inside an episodic section each dated entry is its own unit, which turns a
12
+ * changelog from dead weight into cheap, addressable history.
13
+ *
14
+ * 2. Ranking is BM25 over stemmed tokens, with the unit's heading path and its OnDemandMemory
15
+ * file's list triggers weighted three times the body. The previous router scored one point per
16
+ * trigger whose words all appeared exactly; on five realistic tasks it found a unit for
17
+ * two. This one found all five. It stays inside O4's constraints: no embeddings, no model
18
+ * call, no network, deterministic, ties broken by document order.
19
+ *
20
+ * The stemmer is deliberately crude (a suffix list, not Porter): it only has to make `deploy`,
21
+ * `deploys` and `deployment` one term, and it must never produce different output on different
22
+ * machines. Every parameter is a named constant so the interface doc can quote it.
23
+ */
24
+ import type { Manifest } from "./compile.js";
25
+ import type { MemoryKind } from "./episodic.js";
26
+ /** BM25 term-frequency saturation. */
27
+ export declare const BM25_K1 = 1.2;
28
+ /**
29
+ * BM25 length normalisation. 0.75 is the textbook value; it made a 21-token changelog line
30
+ * outrank the 4,467-token section that actually held the answer on two of five audit tasks.
31
+ * 0.3 keeps some preference for short units without letting length alone decide.
32
+ */
33
+ export declare const BM25_B = 0.3;
34
+ /** Heading-path and trigger terms count this many times a body term. */
35
+ export declare const HEADING_WEIGHT = 3;
36
+ /** Default cap on tokens a recall returns; the agent widens with outline() if it needs more. */
37
+ export declare const RECALL_MAX_TOKENS = 1500;
38
+ /**
39
+ * A unit scoring under this fraction of the best hit is not returned even when the cap has
40
+ * room. The cap is a ceiling, not a target: on the audit corpus a query whose answer was two
41
+ * changelog entries (169 tokens) came back with six (1,470) because the tail fit; at 0.3 the
42
+ * tail still fit (it scored 34 to 40% of the best), so the floor is half the best hit.
43
+ */
44
+ export declare const RECALL_MIN_SCORE_RATIO = 0.5;
45
+ export interface Unit {
46
+ /** the OnDemandMemory file this unit came from */
47
+ onDemandFile: string;
48
+ /** OnDemandMemory file path, relative to the compiled directory */
49
+ file: string;
50
+ /** this unit's own heading, or the first 60 characters of a dated entry */
51
+ heading: string;
52
+ /** heading path from the OnDemandMemory file's top heading down to and including this unit */
53
+ path: string[];
54
+ kind: "section" | "entry";
55
+ /** 1-indexed, inclusive, within the OnDemandMemory file */
56
+ startLine: number;
57
+ endLine: number;
58
+ tokens: number;
59
+ content: string;
60
+ }
61
+ /**
62
+ * Crude suffix stemmer. Short words are left alone: stripping "s" from "bus" or "ed" from
63
+ * "red" does more harm than joining their plurals does good.
64
+ */
65
+ export declare function stem(word: string): string;
66
+ /** Lower-cased, stopword-free, stemmed terms of a piece of text. */
67
+ export declare function tokenize(text: string): string[];
68
+ /**
69
+ * Split one OnDemandMemory file's content into retrieval units. Every heading-delimited section
70
+ * is a unit; a section whose bullets are mostly dated (a changelog, a session log) is split
71
+ * further into one unit per entry, with any continuation lines kept with their entry. A heading
72
+ * with no body is a container, not a unit: its text still reaches the router through its
73
+ * children's heading path.
74
+ */
75
+ export declare function unitsOf(onDemandFile: {
76
+ name: string;
77
+ file: string;
78
+ }, content: string): Unit[];
79
+ interface Doc {
80
+ tf: Map<string, number>;
81
+ len: number;
82
+ }
83
+ export interface SearchIndex {
84
+ units: Unit[];
85
+ docs: Doc[];
86
+ df: Map<string, number>;
87
+ avgLen: number;
88
+ }
89
+ /**
90
+ * Build the ranking index once per manifest. `triggers` are the editable per-OnDemandMemory-file
91
+ * trigger words from the manifest; they join every unit of that file at heading weight, so a
92
+ * person who adds a word to the list steers recall without touching the file text.
93
+ */
94
+ export declare function buildSearchIndex(units: Unit[], triggers?: Map<string, string[]>): SearchIndex;
95
+ export interface RankedUnit {
96
+ unit: Unit;
97
+ score: number;
98
+ }
99
+ /**
100
+ * BM25 over the index. Returns only units with a positive score, best first, ties in document
101
+ * order, so the same query against the same files ranks the same way on every machine.
102
+ */
103
+ export declare function rankUnits(index: SearchIndex, query: string, limit?: number): RankedUnit[];
104
+ /**
105
+ * The best units that fit under a token cap, in rank order, dropping the tail that scores under
106
+ * RECALL_MIN_SCORE_RATIO of the best hit. Always returns the top unit even when it alone
107
+ * exceeds the cap: an answer that is too long is better than no answer, and the caller sees the
108
+ * token count and can decide.
109
+ */
110
+ export declare function selectUnits(ranked: RankedUnit[], maxTokens?: number): RankedUnit[];
111
+ /**
112
+ * Trigger words for an OnDemandMemory file, chosen against the other OnDemandMemory files
113
+ * rather than by raw frequency: a term scores by how often it appears here, times how rare it
114
+ * is across the set, with heading terms weighted. The previous frequency ranking produced
115
+ * `bars, free, merged` for a trading-app memory file (audit 2026-09-04). Returns surface words,
116
+ * not stems, so the list stays readable; two words with one stem count once.
117
+ */
118
+ export declare function selectTriggers(onDemandFiles: {
119
+ name: string;
120
+ heading: string;
121
+ content: string;
122
+ }[], limit?: number): Map<string, string[]>;
123
+ export interface RankedOnDemandFile {
124
+ name: string;
125
+ file: string;
126
+ kind?: MemoryKind;
127
+ triggers: string[];
128
+ tokens: number;
129
+ score: number;
130
+ }
131
+ /**
132
+ * Deterministic keyword-overlap ranking: one point per trigger whose words all appear as whole
133
+ * words in the query. Whole words, not substrings, so the trigger "app" does not match "happy"
134
+ * and "api" does not match "capital". Stable-sorted by score desc, then by manifest order, so the
135
+ * same query always ranks the same way - reproducibility matters more than ranking
136
+ * sophistication here. The older trigger-overlap ranker over the manifest alone, kept for
137
+ * bench.ts's agent-driven-route model (needs no file content, unlike rankUnits above); moved
138
+ * here from recall.ts (2026-09-13 audit round 2, D7c) to sit beside the rest of the ranking code.
139
+ */
140
+ export declare function rankOnDemandFiles(manifest: Manifest, query: string): RankedOnDemandFile[];
141
+ export {};
package/dist/router.js ADDED
@@ -0,0 +1,314 @@
1
+ /**
2
+ * The router: deterministic lexical ranking over section-level units (interface rule O4).
3
+ *
4
+ * Two decisions here, both from the 2026-09-04 interface audit
5
+ * (interface audit, 2026-09-04):
6
+ *
7
+ * 1. The retrieval unit is a section, not an OnDemandMemory file. A file-level hit on a real
8
+ * memory file cost 1,303 to 8,825 tokens, and the largest file was the changelog, so better
9
+ * recall at file granularity opened exactly the content whose non-opening produced the
10
+ * saving. At section granularity the same queries cost 17 to 60 tokens where the file had
11
+ * structure. Inside an episodic section each dated entry is its own unit, which turns a
12
+ * changelog from dead weight into cheap, addressable history.
13
+ *
14
+ * 2. Ranking is BM25 over stemmed tokens, with the unit's heading path and its OnDemandMemory
15
+ * file's list triggers weighted three times the body. The previous router scored one point per
16
+ * trigger whose words all appeared exactly; on five realistic tasks it found a unit for
17
+ * two. This one found all five. It stays inside O4's constraints: no embeddings, no model
18
+ * call, no network, deterministic, ties broken by document order.
19
+ *
20
+ * The stemmer is deliberately crude (a suffix list, not Porter): it only has to make `deploy`,
21
+ * `deploys` and `deployment` one term, and it must never produce different output on different
22
+ * machines. Every parameter is a named constant so the interface doc can quote it.
23
+ */
24
+ import { isEpisodicBody, RE_DATED_BULLET, sections, STOPWORDS, weightedFragments } from "./text.js";
25
+ import { estimateTokens } from "./tokenizer.js";
26
+ /** BM25 term-frequency saturation. */
27
+ export const BM25_K1 = 1.2;
28
+ /**
29
+ * BM25 length normalisation. 0.75 is the textbook value; it made a 21-token changelog line
30
+ * outrank the 4,467-token section that actually held the answer on two of five audit tasks.
31
+ * 0.3 keeps some preference for short units without letting length alone decide.
32
+ */
33
+ export const BM25_B = 0.3;
34
+ /** Heading-path and trigger terms count this many times a body term. */
35
+ export const HEADING_WEIGHT = 3;
36
+ /** Default cap on tokens a recall returns; the agent widens with outline() if it needs more. */
37
+ export const RECALL_MAX_TOKENS = 1500;
38
+ /**
39
+ * A unit scoring under this fraction of the best hit is not returned even when the cap has
40
+ * room. The cap is a ceiling, not a target: on the audit corpus a query whose answer was two
41
+ * changelog entries (169 tokens) came back with six (1,470) because the tail fit; at 0.3 the
42
+ * tail still fit (it scored 34 to 40% of the best), so the floor is half the best hit.
43
+ */
44
+ export const RECALL_MIN_SCORE_RATIO = 0.5;
45
+ const RE_SUFFIX = /(?:ations?|ments?|tions?|ings?|ers?|ed|es|s)$/;
46
+ /**
47
+ * Crude suffix stemmer. Short words are left alone: stripping "s" from "bus" or "ed" from
48
+ * "red" does more harm than joining their plurals does good.
49
+ */
50
+ export function stem(word) {
51
+ if (word.length <= 4)
52
+ return word;
53
+ if (word.endsWith("ies"))
54
+ return `${word.slice(0, -3)}y`;
55
+ const stripped = word.replace(RE_SUFFIX, "");
56
+ return stripped.length >= 3 ? stripped : word;
57
+ }
58
+ /** Lower-cased, stopword-free, stemmed terms of a piece of text. */
59
+ export function tokenize(text) {
60
+ const words = text.toLowerCase().match(/[a-z][a-z0-9_-]{1,}/g) ?? [];
61
+ const out = [];
62
+ for (const w of words) {
63
+ if (STOPWORDS.has(w))
64
+ continue;
65
+ const s = stem(w);
66
+ if (s.length >= 3 && !STOPWORDS.has(s))
67
+ out.push(s);
68
+ }
69
+ return out;
70
+ }
71
+ const RE_BULLET = /^\s*[-*+]\s+/;
72
+ function entryHeading(line) {
73
+ const text = line.replace(RE_BULLET, "").replace(/\*\*/g, "").trim();
74
+ return text.length > 60 ? `${text.slice(0, 57)}...` : text;
75
+ }
76
+ /**
77
+ * Split one OnDemandMemory file's content into retrieval units. Every heading-delimited section
78
+ * is a unit; a section whose bullets are mostly dated (a changelog, a session log) is split
79
+ * further into one unit per entry, with any continuation lines kept with their entry. A heading
80
+ * with no body is a container, not a unit: its text still reaches the router through its
81
+ * children's heading path.
82
+ */
83
+ export function unitsOf(onDemandFile, content) {
84
+ const lines = content.split(/\r?\n/);
85
+ const secs = sections(lines);
86
+ const units = [];
87
+ const stack = [];
88
+ for (const sec of secs) {
89
+ while (stack.length > 0 && (stack[stack.length - 1]?.level ?? 0) >= sec.level)
90
+ stack.pop();
91
+ if (sec.level > 0)
92
+ stack.push({ level: sec.level, heading: sec.heading });
93
+ const path = stack.map((s) => s.heading);
94
+ // sections() gives 1-indexed inclusive start and end; the heading is the first line.
95
+ const bodyStart = sec.level > 0 ? sec.startLine + 1 : sec.startLine;
96
+ const bodyLines = lines.slice(bodyStart - 1, sec.endLine);
97
+ if (bodyLines.every((l) => l.trim() === ""))
98
+ continue;
99
+ const episodic = isEpisodicBody(bodyLines);
100
+ if (!episodic) {
101
+ const text = lines.slice(sec.startLine - 1, sec.endLine).join("\n").trimEnd();
102
+ units.push({
103
+ onDemandFile: onDemandFile.name,
104
+ file: onDemandFile.file,
105
+ heading: sec.heading,
106
+ path,
107
+ kind: "section",
108
+ startLine: sec.startLine,
109
+ endLine: sec.endLine,
110
+ tokens: estimateTokens(text),
111
+ content: text,
112
+ });
113
+ continue;
114
+ }
115
+ // One unit per dated entry. Lines before the first entry (an intro sentence) stay with the
116
+ // section heading as their own unit so nothing is lost.
117
+ let entryStart = -1;
118
+ const flush = (endLine) => {
119
+ if (entryStart < 0)
120
+ return;
121
+ const text = lines.slice(entryStart - 1, endLine).join("\n").trimEnd();
122
+ const heading = entryHeading(lines[entryStart - 1] ?? "");
123
+ units.push({
124
+ onDemandFile: onDemandFile.name,
125
+ file: onDemandFile.file,
126
+ heading,
127
+ path: [...path, heading],
128
+ kind: "entry",
129
+ startLine: entryStart,
130
+ endLine,
131
+ tokens: estimateTokens(text),
132
+ content: text,
133
+ });
134
+ entryStart = -1;
135
+ };
136
+ const introEnd = bodyLines.findIndex((l) => RE_DATED_BULLET.test(l));
137
+ if (introEnd > 0 && bodyLines.slice(0, introEnd).some((l) => l.trim() !== "")) {
138
+ const text = lines.slice(sec.startLine - 1, bodyStart - 1 + introEnd).join("\n").trimEnd();
139
+ units.push({
140
+ onDemandFile: onDemandFile.name,
141
+ file: onDemandFile.file,
142
+ heading: sec.heading,
143
+ path,
144
+ kind: "section",
145
+ startLine: sec.startLine,
146
+ endLine: bodyStart - 1 + introEnd,
147
+ tokens: estimateTokens(text),
148
+ content: text,
149
+ });
150
+ }
151
+ for (let i = 0; i < bodyLines.length; i++) {
152
+ const lineNo = bodyStart + i;
153
+ if (RE_DATED_BULLET.test(bodyLines[i] ?? "")) {
154
+ flush(lineNo - 1);
155
+ entryStart = lineNo;
156
+ }
157
+ }
158
+ flush(sec.endLine);
159
+ }
160
+ return units;
161
+ }
162
+ /**
163
+ * Build the ranking index once per manifest. `triggers` are the editable per-OnDemandMemory-file
164
+ * trigger words from the manifest; they join every unit of that file at heading weight, so a
165
+ * person who adds a word to the list steers recall without touching the file text.
166
+ */
167
+ export function buildSearchIndex(units, triggers = new Map()) {
168
+ const docs = units.map((u) => {
169
+ const tf = new Map();
170
+ const add = (terms, weight) => {
171
+ for (const t of terms)
172
+ tf.set(t, (tf.get(t) ?? 0) + weight);
173
+ };
174
+ add(tokenize(u.path.join(" ")), HEADING_WEIGHT);
175
+ add(tokenize((triggers.get(u.onDemandFile) ?? []).join(" ")), HEADING_WEIGHT);
176
+ // The body without its heading line, which is already counted at heading weight.
177
+ add(tokenize(u.kind === "section" ? u.content.split("\n").slice(1).join("\n") : u.content), 1);
178
+ let len = 0;
179
+ for (const n of tf.values())
180
+ len += n;
181
+ return { tf, len };
182
+ });
183
+ const df = new Map();
184
+ for (const d of docs)
185
+ for (const t of d.tf.keys())
186
+ df.set(t, (df.get(t) ?? 0) + 1);
187
+ const avgLen = docs.length === 0 ? 0 : docs.reduce((n, d) => n + d.len, 0) / docs.length;
188
+ return { units, docs, df, avgLen };
189
+ }
190
+ /**
191
+ * BM25 over the index. Returns only units with a positive score, best first, ties in document
192
+ * order, so the same query against the same files ranks the same way on every machine.
193
+ */
194
+ export function rankUnits(index, query, limit = 20) {
195
+ const terms = [...new Set(tokenize(query))];
196
+ if (terms.length === 0 || index.units.length === 0)
197
+ return [];
198
+ const n = index.units.length;
199
+ const scored = [];
200
+ index.docs.forEach((d, i) => {
201
+ let score = 0;
202
+ for (const t of terms) {
203
+ const f = d.tf.get(t) ?? 0;
204
+ if (f === 0)
205
+ continue;
206
+ const dfT = index.df.get(t) ?? 0;
207
+ const idf = Math.log(1 + (n - dfT + 0.5) / (dfT + 0.5));
208
+ const norm = 1 - BM25_B + (BM25_B * d.len) / (index.avgLen || 1);
209
+ score += (idf * (f * (BM25_K1 + 1))) / (f + BM25_K1 * norm);
210
+ }
211
+ if (score > 0)
212
+ scored.push({ i, score: Math.round(score * 1000) / 1000 });
213
+ });
214
+ return scored
215
+ .sort((a, b) => b.score - a.score || a.i - b.i)
216
+ .slice(0, limit)
217
+ .map(({ i, score }) => ({ unit: index.units[i], score }));
218
+ }
219
+ /**
220
+ * The best units that fit under a token cap, in rank order, dropping the tail that scores under
221
+ * RECALL_MIN_SCORE_RATIO of the best hit. Always returns the top unit even when it alone
222
+ * exceeds the cap: an answer that is too long is better than no answer, and the caller sees the
223
+ * token count and can decide.
224
+ */
225
+ export function selectUnits(ranked, maxTokens = RECALL_MAX_TOKENS) {
226
+ const out = [];
227
+ const floor = (ranked[0]?.score ?? 0) * RECALL_MIN_SCORE_RATIO;
228
+ let used = 0;
229
+ for (const r of ranked) {
230
+ if (out.length > 0 && (r.score < floor || used + r.unit.tokens > maxTokens))
231
+ break;
232
+ out.push(r);
233
+ used += r.unit.tokens;
234
+ }
235
+ return out;
236
+ }
237
+ /**
238
+ * Trigger words for an OnDemandMemory file, chosen against the other OnDemandMemory files
239
+ * rather than by raw frequency: a term scores by how often it appears here, times how rare it
240
+ * is across the set, with heading terms weighted. The previous frequency ranking produced
241
+ * `bars, free, merged` for a trading-app memory file (audit 2026-09-04). Returns surface words,
242
+ * not stems, so the list stays readable; two words with one stem count once.
243
+ */
244
+ export function selectTriggers(onDemandFiles, limit = 6) {
245
+ const surface = new Map();
246
+ const perFile = onDemandFiles.map((m) => {
247
+ const tf = new Map();
248
+ const add = (text, weight) => {
249
+ const words = text.toLowerCase().match(/[a-z][a-z0-9_-]{2,}/g) ?? [];
250
+ for (const w of words) {
251
+ if (STOPWORDS.has(w))
252
+ continue;
253
+ const s = stem(w);
254
+ if (s.length < 3)
255
+ continue;
256
+ tf.set(s, (tf.get(s) ?? 0) + weight);
257
+ // Keep the shortest surface form seen for a stem, which is usually the base word.
258
+ const seen = surface.get(s);
259
+ if (!seen || w.length < seen.length)
260
+ surface.set(s, w);
261
+ }
262
+ };
263
+ add(m.heading, HEADING_WEIGHT);
264
+ for (const { text, weight } of weightedFragments(m.content))
265
+ add(text, weight);
266
+ return tf;
267
+ });
268
+ const df = new Map();
269
+ for (const tf of perFile)
270
+ for (const t of tf.keys())
271
+ df.set(t, (df.get(t) ?? 0) + 1);
272
+ const n = onDemandFiles.length;
273
+ const out = new Map();
274
+ onDemandFiles.forEach((m, i) => {
275
+ const tf = perFile[i];
276
+ const ranked = [...tf.entries()]
277
+ .map(([t, f]) => {
278
+ // Plain idf: a term in every file scores zero and drops out, which is the point. With
279
+ // one or two files idf is degenerate, so every term counts and headings still lead.
280
+ const rarity = n >= 3 ? Math.log(n / (df.get(t) ?? 1)) : 1;
281
+ return { t, score: f * rarity };
282
+ })
283
+ .filter((x) => x.score > 0)
284
+ .sort((a, b) => b.score - a.score || a.t.localeCompare(b.t))
285
+ .slice(0, limit)
286
+ .map(({ t }) => surface.get(t) ?? t);
287
+ out.set(m.name, ranked.length > 0 ? ranked : [m.name.replace(/_/g, " ")]);
288
+ });
289
+ return out;
290
+ }
291
+ /**
292
+ * Deterministic keyword-overlap ranking: one point per trigger whose words all appear as whole
293
+ * words in the query. Whole words, not substrings, so the trigger "app" does not match "happy"
294
+ * and "api" does not match "capital". Stable-sorted by score desc, then by manifest order, so the
295
+ * same query always ranks the same way - reproducibility matters more than ranking
296
+ * sophistication here. The older trigger-overlap ranker over the manifest alone, kept for
297
+ * bench.ts's agent-driven-route model (needs no file content, unlike rankUnits above); moved
298
+ * here from recall.ts (2026-09-13 audit round 2, D7c) to sit beside the rest of the ranking code.
299
+ */
300
+ export function rankOnDemandFiles(manifest, query) {
301
+ const words = new Set(query.toLowerCase().match(/[a-z0-9][a-z0-9_-]*/g) ?? []);
302
+ const matches = (trigger) => trigger
303
+ .toLowerCase()
304
+ .split(/\s+/)
305
+ .filter(Boolean)
306
+ .every((w) => words.has(w));
307
+ return manifest.onDemandFiles
308
+ .map((m, i) => {
309
+ const score = m.triggers.reduce((n, t) => n + (matches(t) ? 1 : 0), 0);
310
+ return { name: m.name, file: m.file, kind: m.kind, triggers: m.triggers, tokens: m.tokens, score, _i: i };
311
+ })
312
+ .sort((a, b) => b.score - a.score || a._i - b._i)
313
+ .map(({ _i, ...rest }) => rest);
314
+ }
@@ -0,0 +1,32 @@
1
+ /**
2
+ * The MM001-MM010 rule set.
3
+ *
4
+ * Every rule is pure: it reads a Workspace and returns Findings. Nothing here touches the
5
+ * filesystem or the network, which is what makes the whole engine testable from fixtures.
6
+ */
7
+ import { GRANULARITY_TOKENS, type Block } from "./text.js";
8
+ import { type Finding, type MemoryFile, type Rule } from "./types.js";
9
+ export interface DuplicateGroup {
10
+ occurrences: {
11
+ file: MemoryFile;
12
+ block: Block;
13
+ }[];
14
+ wastedTokens: number;
15
+ }
16
+ /**
17
+ * Blocks repeated verbatim (modulo list-marker/heading-hash/whitespace normalisation) across two
18
+ * or more files. Shared by MM005 and the multi-file scan (scan.ts) so both see the same picture
19
+ * of what a reorganize plan should merge.
20
+ */
21
+ export declare function findDuplicateBlocks(files: MemoryFile[]): DuplicateGroup[];
22
+ /**
23
+ * The hook is the descriptive text, with the list marker, any leading markdown link, and a
24
+ * separator dash stripped. That is the part a human writes and the part worth budgeting;
25
+ * the link is structural overhead the author cannot shorten much.
26
+ */
27
+ export declare function hookText(line: string): string;
28
+ /** MM008 for one file. Exported so `init` can refuse to copy a credential into new files. */
29
+ export declare function secretFindings(file: MemoryFile): Finding[];
30
+ export { GRANULARITY_TOKENS };
31
+ export declare const RULES: Rule[];
32
+ export declare function ruleById(id: string): Rule | undefined;