@fanchao8609/agent_brain_sync 1.8.8 → 1.8.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,244 @@
1
+ /**
2
+ * 相关页推荐:按"当前在干什么"挑出该读的知识页。
3
+ *
4
+ * 为何需要它(2026-09-16 实测):
5
+ * MCP 日志里 abs_task 234 / abs_note 54 / abs_load 53,而 abs_query 只有 2 次。
6
+ * 写入 288 : 检索 2 = 144:1。经验写进去了,几乎从没被读回来。
7
+ *
8
+ * 根因不是"忘了查",是【给人看目录,人不会去翻书】:
9
+ * load 只给 index 的一句话清单 → 模型觉得"我记过了" → 开工 → 真需要时靠印象。
10
+ * 模型不知道自己不知道什么,所以永远不会主动 query。
11
+ *
12
+ * 解法:不依赖主动查询 —— load 时【被动带出】相关页正文。
13
+ * 匹配依据是"当前在干什么",不是"主题词":
14
+ * · 最近改过的源文件名 → 我现在在动哪个模块
15
+ * · todo 里活跃任务 → 我现在在做什么事
16
+ *
17
+ * 为何不用 git:本项目 store.js 是纯文件操作,引 git 要处理
18
+ * “没装 git / 不是 repo / 子进程 27ms”三种情况,为取几个文件名不值。
19
+ * 改用 mtime 扫最近改动 —— 信息量相同,零依赖。
20
+ *
21
+ * 与 query 的区别:
22
+ * query 需要用户/模型先想到一个词;本模块不需要 —— 输入是环境事实。
23
+ */
24
+
25
+ /** 中文/英文都按"够长才算有效词"处理,避免 the/的/了 这类噪音。 */
26
+ const STOP = new Set([
27
+ 'the', 'and', 'for', 'with', 'this', 'that', 'from', 'into', 'when', 'then',
28
+ 'abs', 'src', 'test', 'js', 'ts', 'md', 'json', 'node',
29
+ '的', '了', '是', '在', '和', '与', '或', '把', '被', '给', '对', '从',
30
+ ]);
31
+
32
+ /** 从任意文本抽关键词:英文词(>=3) + 中文 2-gram。 */
33
+ export function keywords(text) {
34
+ const out = new Set();
35
+ const s = String(text || '');
36
+ // 英文/数字/下划线词
37
+ for (const m of s.matchAll(/[A-Za-z][A-Za-z0-9_-]{2,}/g)) {
38
+ const w = m[0].toLowerCase();
39
+ if (!STOP.has(w)) out.add(w);
40
+ }
41
+ // 中文串 → 2-gram(中文没有词边界,2-gram 是最省事的可匹配单位)
42
+ for (const m of s.matchAll(/[\u4e00-\u9fa5]{2,}/g)) {
43
+ const seg = m[0];
44
+ for (let i = 0; i + 2 <= seg.length; i++) {
45
+ const g = seg.slice(i, i + 2);
46
+ if (!STOP.has(g)) out.add(g);
47
+ }
48
+ }
49
+ return [...out];
50
+ }
51
+
52
+ /**
53
+ * 给一页打分:命中多少关键词(标题权重 3,正文权重 1)。
54
+ * 不做 TF-IDF —— 图谱就 30 页,朴素加权够用,也更好解释。
55
+ */
56
+ export function scorePage(body, pageName, kws) {
57
+ const title = String(pageName).toLowerCase();
58
+ const text = String(body).toLowerCase();
59
+ let score = 0;
60
+ const hit = [];
61
+ for (const k of kws) {
62
+ const inTitle = title.includes(k);
63
+ const n = text.split(k).length - 1;
64
+ if (!inTitle && !n) continue;
65
+ score += (inTitle ? 3 : 0) + Math.min(n, 3);
66
+ hit.push(k);
67
+ }
68
+ return { score, hit };
69
+ }
70
+
71
+ /**
72
+ * 挑出 top-N 相关页。
73
+ * @param {{name:string,body:string}[]} pages
74
+ * @param {string[]} kws
75
+ * @param {number} n
76
+ */
77
+ export function pickRelevant(pages, kws, n = 3) {
78
+ if (!kws.length) return [];
79
+ return pages
80
+ .map((p) => {
81
+ const { score, hit } = scorePage(p.body, p.name, kws);
82
+ return { ...p, score, hit };
83
+ })
84
+ .filter((p) => p.score > 0)
85
+ // 同分时按名字排,保证输出稳定(否则测试会抖)
86
+ .sort((a, b) => b.score - a.score || a.name.localeCompare(b.name))
87
+ .slice(0, n);
88
+ }
89
+
90
+ /**
91
+ * 扫最近改过的源文件名(mtime 排序,不调 git)。
92
+ * 跳过 .brain/(那是图谱自己)、node_modules、隐藏目录。
93
+ */
94
+ export async function recentFiles(root, { n = 8, days = 3 } = {}) {
95
+ const fs = await import('node:fs/promises');
96
+ const { join } = await import('node:path');
97
+ const cutoff = Date.now() - days * 86400_000;
98
+ const found = [];
99
+ const SKIP = new Set(['node_modules', '.git', '.brain', 'dist', 'build']);
100
+ async function walk(d, depth) {
101
+ if (depth > 3 || found.length > 400) return;
102
+ let ents;
103
+ try { ents = await fs.readdir(d, { withFileTypes: true }); } catch { return; }
104
+ for (const e of ents) {
105
+ if (e.name.startsWith('.') || SKIP.has(e.name)) continue;
106
+ const p = join(d, e.name);
107
+ if (e.isDirectory()) { await walk(p, depth + 1); continue; }
108
+ if (!/\.(js|ts|mjs|cjs|jsx|tsx|py|go|rs|sh)$/.test(e.name)) continue;
109
+ try {
110
+ const st = await fs.stat(p);
111
+ if (st.mtimeMs >= cutoff) found.push({ name: e.name, path: p.replace(root + '/', ''), mtime: st.mtimeMs });
112
+ } catch { /* 忽略不可读 */ }
113
+ }
114
+ }
115
+ await walk(root, 0);
116
+ return found.sort((a, b) => b.mtime - a.mtime).slice(0, n);
117
+ }
118
+
119
+ /**
120
+ * 两级匹配:
121
+ * exact = 查询词直接出现(子串)
122
+ * fuzzy = 查询词的 2-gram 有重叠(如查 "并发写" 能中写"互斥/锁"的页)
123
+ *
124
+ * 为何需要 fuzzy(2026-09-16 实测根因):
125
+ * 原 cmdQuery 只做 body.includes(word) —— 查"并发写"时页里写的是"互斥"、
126
+ * 查"发布流程"时页名是 npm-publish-flow,全部匹配不上。
127
+ * 用户看到"无命中"会以为图谱里没这条经验(静默失效)。
128
+ */
129
+ export function matchPage(body, pageName, queryWords) {
130
+ const text = String(body).toLowerCase();
131
+ const name = String(pageName).toLowerCase();
132
+ const exact = [];
133
+ for (const w of queryWords) {
134
+ const q = w.toLowerCase();
135
+ if (text.includes(q) || name.includes(q)) exact.push(w);
136
+ }
137
+ // 模糊:拿查询词的字符 2-gram 去页里找,重叠度足够就算关联
138
+ // 坑(2026-09-16 实测): 曾用 overlap/qGrams.size >= 0.5 —— 太松。
139
+ // 查"zzzz不存在"时 qGrams={zz,不存,存在}, 页里恰好有"存在" → ratio=0.5 误命中。
140
+ // 两个修正: (1) 重复字符的 gram 不算(zz 去重); (2) 至少得命中 2 个不同 gram。
141
+ const qGrams = new Set(queryWords.flatMap((w) => ngrams(String(w).toLowerCase())));
142
+ const pGrams = new Set(ngrams(text.slice(0, 4000)).concat(ngrams(name)));
143
+ let overlap = 0;
144
+ for (const g of qGrams) if (pGrams.has(g)) overlap++;
145
+ const ratio = qGrams.size ? overlap / qGrams.size : 0;
146
+ return { exact, overlap, ratio };
147
+ }
148
+
149
+ /** 拆字符 2-gram(中英文都适用)—— 模糊匹配的最小单位。
150
+ * 先过 [\p{L}\p{N}] 再切,单一字符组成的 gram(如 zz) 没区分度,丢掉。 */
151
+ function ngrams(s) {
152
+ const out = [];
153
+ const clean = s.replace(/[^\p{L}\p{N}]+/gu, '');
154
+ for (let i = 0; i + 2 <= clean.length; i++) {
155
+ const g = clean.slice(i, i + 2);
156
+ if (g[0] === g[1]) continue; // zz / 11 这类无信息量
157
+ out.push(g);
158
+ }
159
+ return out;
160
+ }
161
+
162
+ /** 模糊门槛 + 最小 gram 数(见 rankPage 注释) */
163
+ export const FUZZY_MIN = 0.6;
164
+ // 查询词去重后的 gram 数必须≥4:否则 "zzzz不存在" 只剩 [不存,存在] 两个 gram,
165
+ // 两个都在任意中文页里 → ratio=100% 误命中(实测)。分母够大才拉得动比例。
166
+ export const FUZZY_MIN_GRAMS = 4;
167
+
168
+ /** 从 frontmatter 取 tags 列表。无 frontmatter / 无 tags → []。 */
169
+ export function tagsOf(body) {
170
+ const m = String(body || '').match(/^tags:\s*(.+)$/m);
171
+ if (!m) return [];
172
+ return m[1]
173
+ .replace(/^\[|\]$/g, '')
174
+ .split(',')
175
+ .map((t) => t.trim())
176
+ .filter(Boolean);
177
+ }
178
+
179
+ /**
180
+ * 端到端检索:把一组查询词打成一页的分数。
181
+ * 返回 null = 不相关(既不精确命中,模糊重叠也不够)。
182
+ *
183
+ * 权重(tags 最高,2026-09-16 用户定):
184
+ * tags 命中 —— 人工提炼的关键词,最可靠 → 权重 8/个
185
+ * 标题命中 —— 页名往往就是主题 → 权重 4/个
186
+ * 正文命中 —— 可能出现但可能只是提一句 → 权重 1/个(封顶 3)
187
+ *
188
+ * 为何 tags 优先:实测现有页的 tags 写着大量人工词组("静默失效"、
189
+ * "双副本"、"陈旧路径"),而 firstHitLine 一直把 tags 行当噪音跳过 ——
190
+ * 等于把最准的检索信号丢掉了。
191
+ */
192
+ export function rankPage(body, pageName, queryWords) {
193
+ const { exact, overlap, ratio } = matchPage(body, pageName, queryWords);
194
+ if (exact.length) {
195
+ const tags = tagsOf(body).map((t) => t.toLowerCase());
196
+ const name = String(pageName).toLowerCase();
197
+ const text = String(body).toLowerCase();
198
+ let score = 10 + exact.length * 5;
199
+ const via = { tag: [], title: [], body: [] };
200
+ for (const w of exact) {
201
+ const q = String(w).toLowerCase();
202
+ if (tags.some((t) => t === q || t.includes(q) || q.includes(t))) {
203
+ score += 8; via.tag.push(w);
204
+ } else if (name.includes(q)) {
205
+ score += 4; via.title.push(w);
206
+ } else if (text.includes(q)) {
207
+ score += 1; via.body.push(w);
208
+ }
209
+ }
210
+ return { kind: 'exact', matched: exact, score, overlap, ratio, via };
211
+ }
212
+ const qGramCount = new Set(queryWords.flatMap((w) => ngrams(String(w).toLowerCase()))).size;
213
+ if (qGramCount >= FUZZY_MIN_GRAMS && ratio >= FUZZY_MIN)
214
+ return { kind: 'fuzzy', matched: [], score: ratio * 10, overlap, ratio, via: { tag: [], title: [], body: [] } };
215
+ return null;
216
+ }
217
+
218
+ /** 取页正文的摘要:优先 frontmatter 的 description,退化到首个非标题行。 */
219
+ export function digest(body, maxLen = 200) {
220
+ const lines = String(body || '').split('\n');
221
+ const fmEnd = lines[0] === '---' ? lines.indexOf('---', 1) : -1;
222
+ if (fmEnd > 0) {
223
+ const d = lines.slice(1, fmEnd).find((l) => /^description\s*:/.test(l));
224
+ if (d) return d.replace(/^description\s*:\s*/, '').trim().slice(0, maxLen);
225
+ }
226
+ // 无 description 时跳过整个 frontmatter 区 —— 否则会把 'title: X' 当正文。
227
+ const start = fmEnd > 0 ? fmEnd + 1 : 0;
228
+ const first = lines
229
+ .slice(start)
230
+ .find((l) => l.trim() && !l.startsWith('#') && !l.startsWith('---'));
231
+ return (first || '').trim().slice(0, maxLen);
232
+ }
233
+
234
+ /** 渲染"该读的页"段。无命中返回空串(不占 load 体积)。 */
235
+ export function renderRelevant(picked, why = '') {
236
+ if (!picked.length) return '';
237
+ const out = [`--- 相关页(据当前改动/任务自动带出)${why ? ` [${why}]` : ''} ---`];
238
+ for (const p of picked) {
239
+ out.push(`[[${p.name}]] (${p.score}) — ${digest(p.body)}`);
240
+ out.push(` 命中: ${p.hit.slice(0, 8).join(' ')}`);
241
+ }
242
+ out.push('(这些页与你当前在做的事相关;不必再 query)');
243
+ return out.join('\n');
244
+ }