@yolk_vat-y/dsh-project-memory 0.5.4 → 0.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,468 @@
1
+ /**
2
+ * 就绪引擎(push 侧)。
3
+ *
4
+ * recall 回答"记忆里有什么和这件事有关";readiness 回答"动手之前必须知道什么"。
5
+ * 交付契约(设计核心):
6
+ * - **authored trigger = 确定性注入**:作者写下的 keywords / symbols / actions / paths 命中即在
7
+ * 动手前全文注入,优先级 1。可审计(返回 `why`),不参与任何相似度打分。
8
+ * - **统计信号 = 提示态**:没有 trigger 的条目只能作为截断提示进入,优先级 2,且判据是
9
+ * **层内相对阈值**(相对最高分),不是长度敏感的魔法常量。
10
+ *
11
+ * 两个窗口(忠于宿主契约:DSH 没有"工具执行前拦截"钩子,唯一能加上下文的缝是 agent/pre-step):
12
+ * - 先发窗口:人类消息里的意图词("提交/发布/公开")→ 动作 id 在模型规划动作之前就命中;
13
+ * - 反应窗口:已观察到的 tool/call 参数(`git commit`、`npm publish`、改动的路径)走同一个匹配器。
14
+ *
15
+ * 纯函数、零依赖、不调用模型。
16
+ * @module dsh-project-memory/readiness
17
+ */
18
+
19
+ import { FALLBACK_OPS, WEAK_OPS, activityFromText, opForLegacyAction } from './ops.js'
20
+ import { CJK_RANGE, tokenize } from './util/search.js'
21
+
22
+ /**
23
+ * 动作词典:动作 id → 匹配模式。命令形态与人类意图词都归一成同一套 id,
24
+ * 于是"你顺便提交一下"与 `git commit -m x` 对 trigger 是同一件事。
25
+ */
26
+ export const ACTION_LEXICON = [
27
+ ['git-add', [/git\s+add\b/i]],
28
+ ['git-commit', [/git\s+commit\b/i, /提交/]],
29
+ ['git-push', [/git\s+push\b/i, /推送/]],
30
+ ['npm-publish', [/npm\s+publish\b/i, /发包/, /发布\s*到\s*npm/i]],
31
+ ['release', [/git\s+tag\b/i, /\brelease\b/i, /发版/, /发布版本/]],
32
+ ['go-public', [/公开/, /开源/, /对外/, /上线/]],
33
+ ['delete', [/rm\s+-rf\b/, /git\s+rm\b/, /删除/, /清理/]],
34
+ ['migrate', [/migrate\b/i, /迁移/]],
35
+ ['deploy', [/\bdeploy\b/i, /部署/]],
36
+ ]
37
+
38
+ /** 带扩展名的路径 token(README.md、src/util/fs.js …)。 */
39
+ const PATH_TOKEN = /(?:[A-Za-z0-9_.@-]+\/)*[A-Za-z0-9_.@-]+\.[A-Za-z][A-Za-z0-9]{0,7}\b/g
40
+ /** 点开头的裸文件名(.gitignore、.npmrc …)。 */
41
+ const DOTFILE_TOKEN = /(?:^|[\s"'`(])(\.[A-Za-z0-9_-]{2,})/g
42
+
43
+ /**
44
+ * 文本 → 动作 id 列表(确定性、去重、词典序稳定)。
45
+ * @param {string} text - 人类消息或工具调用参数。
46
+ * @returns {string[]} 动作 id。
47
+ */
48
+ export function detectActions(text) {
49
+ const s = String(text || '')
50
+ if (!s) return []
51
+ const out = []
52
+ for (const [id, patterns] of ACTION_LEXICON) {
53
+ if (patterns.some((re) => re.test(s))) out.push(id)
54
+ }
55
+ return out
56
+ }
57
+
58
+ /**
59
+ * 文本 → 路径 token(用于 trigger.paths 的 glob 匹配)。
60
+ * @param {string} text - 工具调用参数或人类消息。
61
+ * @returns {string[]} 去重后的路径 token。
62
+ */
63
+ export function extractPaths(text) {
64
+ const s = String(text || '')
65
+ if (!s) return []
66
+ const out = new Set()
67
+ for (const m of s.matchAll(PATH_TOKEN)) out.add(m[0])
68
+ for (const m of s.matchAll(DOTFILE_TOKEN)) out.add(m[1])
69
+ return [...out]
70
+ }
71
+
72
+ /**
73
+ * 组装就绪上下文:人类意图 + 已观察动作。三处来源都做动作/路径归一,调用方只需给文本。
74
+ * @param {object} input
75
+ * @param {string} [input.humanText] - 本轮人类消息(先发窗口)。
76
+ * @param {string} [input.actionText] - 已观察到的 tool/call 参数(反应窗口)。
77
+ * @param {string[]} [input.actions] - 调用方已归一化的动作 id(可选,会被并集)。
78
+ * @param {string[]} [input.paths] - 调用方已归一化的路径(可选,会被并集)。
79
+ * @returns {{ humanText: string, actionText: string, actions: string[], paths: string[] }}
80
+ */
81
+ export function buildReadinessContext(input = {}) {
82
+ const humanText = String(input.humanText ?? input.query ?? '')
83
+ const actionText = String(input.actionText || '')
84
+ const actions = new Set([...(input.actions || []), ...detectActions(humanText), ...detectActions(actionText)])
85
+ const paths = new Set([...(input.paths || []), ...extractPaths(actionText), ...extractPaths(humanText)])
86
+ // 动作平面(S1):结构化调用优先;没有结构化调用时对 actionText 跑一遍 shell 规则兜底。
87
+ const fromText = activityFromText(actionText)
88
+ const ops = new Set([...(input.ops || []), ...fromText.ops])
89
+ for (const a of actions) {
90
+ const op = opForLegacyAction(a)
91
+ if (op) ops.add(op)
92
+ }
93
+ const targets = new Set([...(input.targets || []), ...fromText.targets])
94
+ const hosts = new Set([...(input.hosts || []), ...fromText.hosts])
95
+ return {
96
+ humanText,
97
+ actionText,
98
+ actions: [...actions],
99
+ paths: [...paths],
100
+ ops: [...ops],
101
+ targets: [...targets],
102
+ hosts: [...hosts],
103
+ intent: intentText(humanText),
104
+ tags: Array.isArray(input.tags) ? [...input.tags] : [],
105
+ }
106
+ }
107
+
108
+ /**
109
+ * 派生 trigger 的意图词表(有界、确定性)。派生信号**只进提示通道**:它能提升召回,
110
+ * 但永不强制注入——"自动学出来的东西"不该污染上下文。
111
+ */
112
+ export const ACTION_WORDS = [
113
+ '提交', '公开', '泄漏', '发布', '发包', '发版', '上线', '部署', '迁移', '删除', '清理', '回滚',
114
+ '推送', '打包', '构建', '依赖', '密钥', '权限', '并发', '超时', '内存', '性能', '安全', '面试',
115
+ ]
116
+
117
+ /** 无扩展名但明确是文件名的裸词(README / CHANGELOG / …),当关键词用。 */
118
+ const BARE_FILE_NAMES = /\b(?:README|CHANGELOG|LICENSE|AGENTS|CONTRIBUTING|Dockerfile|Makefile)\b/g
119
+
120
+ const DERIVED_MAX_KEYWORDS = 6
121
+ const DERIVED_MAX_ACTIONS = 4
122
+ const DERIVED_MAX_PATHS = 6
123
+
124
+ /**
125
+ * 从一条 insight 的正文确定性派生触发信号(零模型、可重放)。
126
+ * 结果只写进 `triggerDerived`,供提示通道(检索文本)使用;是否强制注入只由 authored `trigger` 决定。
127
+ * @param {object} ins - 归一化后的 insight 条目。
128
+ * @returns {{ keywords: string[], actions: string[], paths: string[] }}
129
+ */
130
+ export function deriveTrigger(ins) {
131
+ const text = [
132
+ ins?.title,
133
+ ins?.pattern,
134
+ ins?.fix,
135
+ ins?.choice,
136
+ ins?.reason,
137
+ ins?.problem,
138
+ ins?.solution,
139
+ ins?.topic,
140
+ ins?.body,
141
+ ...(Array.isArray(ins?.steps) ? ins.steps : []),
142
+ ].filter(Boolean).join('\n')
143
+ const bare = text.match(BARE_FILE_NAMES) || []
144
+ const keywords = [...new Set([...ACTION_WORDS.filter((w) => text.includes(w)), ...bare])].slice(0, DERIVED_MAX_KEYWORDS)
145
+ const actions = detectActions(text).slice(0, DERIVED_MAX_ACTIONS)
146
+ const paths = extractPaths(text).slice(0, DERIVED_MAX_PATHS)
147
+ return { keywords, actions, paths }
148
+ }
149
+
150
+ /**
151
+ * insights 文档的懒回填:给缺少 `triggerDerived` 的条目补上派生信号(增量、幂等)。
152
+ * 只改内存;是否落盘由调用方的 commit 决定。
153
+ * @param {{ items?: object[] }} doc - insights 文档。
154
+ * @returns {boolean} 是否发生了变更(用于置 dirty)。
155
+ */
156
+ export function backfillDerivedTriggers(doc) {
157
+ if (!doc || !Array.isArray(doc.items)) return false
158
+ let changed = false
159
+ for (const it of doc.items) {
160
+ if (!it || it.triggerDerived) continue
161
+ const derived = deriveTrigger(it)
162
+ if (!derived.keywords.length && !derived.actions.length && !derived.paths.length) continue
163
+ it.triggerDerived = derived
164
+ changed = true
165
+ }
166
+ return changed
167
+ }
168
+
169
+ /**
170
+ * 提示通道的查询文本:剔除 1–3 个字符的拉丁 token。
171
+ *
172
+ * 为什么:`PR` / `CI` / `OS` / `WSL` / `npm` 这类缩写太短、歧义太大,一个巧合命中就能当上
173
+ * 该层最高分,于是以 `relative:1.00` 混进上下文(实测:人类消息里的 "PR" 把一条 task-tools
174
+ * 越权 lesson 顶到了提示位;"wsl" 又把一条 pnpm lesson 顶到了"改文件时间戳"的任务里)。
175
+ * 内容词(≥4 字符的拉丁标识符、CJK 词)不受影响;authored trigger 通道完全不走这里。
176
+ * @param {string} text - 人类消息或工具参数。
177
+ * @returns {string} 过滤后的查询文本。
178
+ */
179
+ export function hintQueryText(text) {
180
+ return String(text || '')
181
+ .split(/\s+/)
182
+ // 路径形态的 token 原样保留(src/util/fs.js 里的 fs / js 是有效证据),
183
+ // 只对独立词做缩写剔除。
184
+ .map((w) => (/[/.]/.test(w) ? w : w.replace(/\b[A-Za-z0-9_]{1,3}\b/g, ' ')))
185
+ .join(' ')
186
+ .replace(/\s+/g, ' ')
187
+ .trim()
188
+ }
189
+
190
+ /** glob(只支持 `*`)→ 正则。 */
191
+ function globToRegExp(pattern) {
192
+ const escaped = String(pattern).replace(/[.+^${}()|[\]\\]/g, '\\$&').replace(/\*/g, '.*')
193
+ return new RegExp(`^${escaped}$`, 'i')
194
+ }
195
+
196
+ /**
197
+ * 剥离**引用内容**:代码块、行内代码、引号内的路径/文件名、裸路径。
198
+ *
199
+ * 为什么必须剥:人类消息里的文件名是**数据**,不是意图。实测中"石啸天-LLM记忆方向调研.pptx"
200
+ * 这个文件名让一条"做调研要先扫 curated 列表"的经验在改文件时间戳的任务里被注入。
201
+ * @param {string} text - 人类消息原文
202
+ * @returns {string} 只保留意图文字的版本
203
+ */
204
+ export function intentText(text) {
205
+ return String(text || '')
206
+ .replace(/```[\s\S]*?```/g, ' ')
207
+ .replace(/`[^`]*`/g, ' ')
208
+ .replace(/"[^"\n]*"/g, ' ')
209
+ .replace(/“[^”\n]*”/g, ' ')
210
+ .replace(/[A-Za-z]:\\[^\s"']*/g, ' ')
211
+ .replace(/(?:[A-Za-z0-9_.@-]+\/)+[A-Za-z0-9_.@-]+/g, ' ')
212
+ // 仓库里的裸文件名(README / CHANGELOG …)也是语料,不是意图
213
+ .replace(/\b(?:README|CHANGELOG|LICENSE|AGENTS|CONTRIBUTING|Dockerfile|Makefile)\b/g, ' ')
214
+ .replace(/\s+/g, ' ')
215
+ .trim()
216
+ }
217
+
218
+ /**
219
+ * 意图词是否值得作为触发信号(S2 的准入过滤)。
220
+ *
221
+ * 规则来自实测:`rm` 命中 `dcterms`、`ppt` 命中 `pptx`、`ms` 命中任何含 ms 的词——短拉丁词
222
+ * 是假阳性制造机;含 `_`/`.`/`/` 的是标识符或路径,属于语料平面,不是意图。
223
+ */
224
+ export function isIntentWord(word) {
225
+ const s = String(word || '').trim()
226
+ if (!s) return false
227
+ if (CJK_RANGE.test(s)) return s.length >= 2
228
+ if (s.length < 5) return false
229
+ if (/[_./\\*]/.test(s)) return false
230
+ if (/\s/.test(s)) return s.length >= 8
231
+ return true
232
+ }
233
+
234
+ const GLOB_RE = /\*/
235
+ /** 扩展名 glob(`*.pptx`)与泛名 glob(`README*`):只能撒谎,不能收窄。 */
236
+ export function isDroppableGlob(pattern) {
237
+ const p = String(pattern || '')
238
+ if (!GLOB_RE.test(p)) return false
239
+ if (/^\*\.\w+$/.test(p)) return true
240
+ if (/^[A-Za-z][A-Za-z0-9_-]*\*$/.test(p)) return true
241
+ return false
242
+ }
243
+
244
+ /**
245
+ * 意图词命中:CJK 用包含,拉丁用词边界(避免 `ppt` 命中 `pptx`)。
246
+ * @param {string} word
247
+ * @param {string} text - 已经是 {@link intentText} 处理过的意图文本
248
+ */
249
+ export function matchIntent(word, text) {
250
+ const w = String(word || '').trim()
251
+ const hay = String(text || '')
252
+ if (!w || !hay) return false
253
+ if (CJK_RANGE.test(w)) return w.length >= 2 && hay.includes(w)
254
+ if (w.length < 4) return false
255
+ const esc = w.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
256
+ return new RegExp(`(^|[^A-Za-z0-9_])${esc}([^A-Za-z0-9_]|$)`, 'i').test(hay)
257
+ }
258
+
259
+ function matchAnyPath(pattern, targets) {
260
+ const p = String(pattern || '')
261
+ if (!p) return false
262
+ const re = globToRegExp(p)
263
+ return (targets || []).some((cand) => {
264
+ const c = String(cand || '')
265
+ if (!c) return false
266
+ return re.test(c) || re.test(c.split(/[\\/]/).pop() || '')
267
+ })
268
+ }
269
+
270
+ /** guard 是**收窄**条件:全部满足才放行;它自己不能触发任何东西。 */
271
+ function guardPass(guard, ctx) {
272
+ if (!guard || typeof guard !== 'object') return true
273
+ const targets = ctx?.targets || []
274
+ if (Array.isArray(guard.paths) && guard.paths.length && !guard.paths.some((p) => matchAnyPath(p, targets))) return false
275
+ if (Array.isArray(guard.not_paths) && guard.not_paths.some((p) => matchAnyPath(p, targets))) return false
276
+ if (Array.isArray(guard.hosts) && guard.hosts.length && !guard.hosts.some((h) => (ctx?.hosts || []).includes(String(h)))) return false
277
+ if (Array.isArray(guard.tags) && guard.tags.length) {
278
+ const tags = ctx?.tags || []
279
+ // 画像未知(tags 为空)时不过滤:这是既有语义,保持兼容。
280
+ if (tags.length && !guard.tags.some((t) => tags.includes(String(t)))) return false
281
+ }
282
+ return true
283
+ }
284
+
285
+ /**
286
+ * trigger 命中判定(确定性,无打分)——**准入化后的语义**。
287
+ *
288
+ * `when` 是唯一的触发面,三个成员按 OR:`ops`(动作)/`writes`(本次要写的文件)/
289
+ * `intents`(人类消息里剥离引用后的意图词)。`guard` 只能收窄。
290
+ * **没有 `when` 的条目不再触发任何东西**(降级为可发现 + 按需拉取)。
291
+ *
292
+ * @param {object|null} trigger
293
+ * @param {object} ctx - {@link buildReadinessContext} 的结果
294
+ * @returns {string|null} 命中原因(`op:…` / `write:…` / `intent:…`),未命中为 null
295
+ */
296
+ export function matchTrigger(trigger, ctx) {
297
+ if (!trigger || typeof trigger !== 'object') return null
298
+ const when = trigger.when
299
+ if (!when || typeof when !== 'object') return null
300
+ if (!guardPass(trigger.guard, ctx)) return null
301
+ for (const op of when.ops || []) {
302
+ if (op && (ctx?.ops || []).includes(String(op))) return `op:${op}`
303
+ }
304
+ for (const w of when.writes || []) {
305
+ // 扩展名/泛名 glob 在这里被硬性忽略:`*.pptx` 这类条件只能撒谎,不能收窄。
306
+ if (!w || isDroppableGlob(w)) continue
307
+ if (matchAnyPath(w, ctx?.targets)) return `write:${w}`
308
+ }
309
+ const intent = ctx?.intent ?? ctx?.humanText ?? ''
310
+ for (const it of when.intents || []) {
311
+ if (it && matchIntent(it, intent)) return `intent:${it}`
312
+ }
313
+ return null
314
+ }
315
+
316
+ /**
317
+ * 旧 trigger → 新 schema(纯函数、幂等、不改原对象)。
318
+ *
319
+ * 迁移规则(都在实测里有据):
320
+ * - `actions` → `when.ops`(经 {@link opForLegacyAction} 映射;死值记录进 `triggerNormalized.dropped`)
321
+ * - 具体的 `paths` → `when.writes`("我要改这个文件");扩展名/泛名 glob 直接丢弃
322
+ * - `keywords` → `when.intents`,只留通过 {@link isIntentWord} 的
323
+ * - `scope` → 默认忽略(它的值不在项目画像 tag 空间里,实测把最相关的一条 procedure 判了死刑)
324
+ *
325
+ * @param {object} it
326
+ * @param {{legacyScope?: 'ignore'|'filter'}} [opts]
327
+ */
328
+ export function normalizeTrigger(it, opts = {}) {
329
+ const t = it && it.trigger
330
+ if (!t || typeof t !== 'object') return it
331
+ if (t.when && typeof t.when === 'object') return it // 已是新 schema:幂等
332
+ const dropped = []
333
+ const writes = []
334
+ for (const p of t.paths || []) {
335
+ if (!p) continue
336
+ if (isDroppableGlob(p)) dropped.push(`path:${p}`)
337
+ else writes.push(String(p))
338
+ }
339
+ const ops = new Set()
340
+ for (const a of t.actions || []) {
341
+ const op = opForLegacyAction(a)
342
+ if (!op) {
343
+ dropped.push(`action:${a}`)
344
+ continue
345
+ }
346
+ // 有精确 writes 时丢掉弱 op:留着它只会把"改这个文件时"扩大成"任何一次提交时"。
347
+ if (writes.length && WEAK_OPS.has(op)) {
348
+ dropped.push(`weak-action:${a}`)
349
+ continue
350
+ }
351
+ ops.add(op)
352
+ }
353
+ // 具体路径只写进 `when.writes`,**不要**再补一个 `file-write` 到 ops:
354
+ // when 的成员是取或的,补进去等于让"写了任何文件"就命中,把路径条件短路掉
355
+ // (实测:那样会让 D 场景一次多出 6 条假阳性)。
356
+ const intents = []
357
+ for (const k of t.keywords || []) if (isIntentWord(k)) intents.push(String(k))
358
+ const when = {}
359
+ if (ops.size) when.ops = [...ops].sort()
360
+ if (writes.length) when.writes = [...new Set(writes)]
361
+ if (intents.length) when.intents = [...new Set(intents)]
362
+ const out = { ...t, when }
363
+ const scopeIgnored = Array.isArray(t.scope) && t.scope.length > 0
364
+ if (scopeIgnored && opts.legacyScope !== 'ignore') out.guard = { ...(t.guard || {}), tags: t.scope }
365
+ return { ...it, trigger: out, triggerNormalized: { dropped, scopeIgnored } }
366
+ }
367
+
368
+ /**
369
+ * trigger 自检(S5):把"哪些条目其实推不动、哪些声明是死的"变成可数的事实。
370
+ * @param {object[]} items
371
+ */
372
+ export function auditTriggers(items, opts = {}) {
373
+ const out = {
374
+ total: 0,
375
+ pushable: 0,
376
+ pullOnly: [],
377
+ deadActions: [],
378
+ droppedGlobs: [],
379
+ weakDropped: [],
380
+ fallbackOps: [],
381
+ scopeIgnored: [],
382
+ missingPrevents: [],
383
+ }
384
+ for (const raw of items || []) {
385
+ if (!raw || !raw.id) continue
386
+ out.total++
387
+ const it = normalizeTrigger(raw, opts)
388
+ const meta = it.triggerNormalized || {}
389
+ for (const d of meta.dropped || []) {
390
+ if (d.startsWith('action:')) out.deadActions.push({ id: it.id, value: d.slice(7) })
391
+ else if (d.startsWith('path:')) out.droppedGlobs.push({ id: it.id, value: d.slice(5) })
392
+ else if (d.startsWith('weak-action:')) out.weakDropped.push({ id: it.id, value: d.slice(12) })
393
+ }
394
+ if (meta.scopeIgnored) out.scopeIgnored.push(it.id)
395
+ const when = it.trigger?.when
396
+ for (const w of when?.writes || []) {
397
+ if (isDroppableGlob(w)) out.droppedGlobs.push({ id: it.id, value: w })
398
+ }
399
+ const hasWhen = Boolean(when && ((when.ops || []).length || (when.writes || []).length || (when.intents || []).length))
400
+ if (hasWhen) {
401
+ out.pushable++
402
+ // 兜底 op 出现在 when.ops 里:不是错误,但这条触发面比它看起来宽得多。
403
+ const fb = (when.ops || []).filter((o) => FALLBACK_OPS.has(o))
404
+ if (fb.length) out.fallbackOps.push({ id: it.id, ops: fb })
405
+ } else {
406
+ out.pullOnly.push(it.id)
407
+ }
408
+ if (!it.trigger?.prevents) out.missingPrevents.push(it.id)
409
+ }
410
+ return out
411
+ }
412
+
413
+ /**
414
+ * IDF 加权覆盖率(S4 的**绝对**门槛):条目覆盖了查询里多少**信息量**,而不是多少个 token。
415
+ *
416
+ * 为什么需要它:`relativeHits` 只看"层内最高分的比例",而最高分本身可能就是噪声——
417
+ * 实测 `relative:1.00` 出现在和查询毫无关系的条目上。归一在 [0,1]、有真零点,才能设下限。
418
+ *
419
+ * 返回四个量,调用方要一起看:
420
+ * - `coverage`:在**语料能表示**的词里,条目覆盖了多少信息量(分母不含 df=0 的词——
421
+ * 自然语言查询总有语料没有的词,把它们算成未命中会让任何正常查询都趋零)。
422
+ * - `matched`:命中的词数。单个通用词("插件")也能拿到 coverage=1.00。
423
+ * - `supported` / `terms`:语料里有对应词的比例。长句子里只有一两个词能在语料中找到对应时,
424
+ * "覆盖率 1.00"是假象(实测:一句 19 个词的改时间戳请求,只与 WSL 笔记共享一个"文件",
425
+ * 却拿到 cov=1.00),调用方据此让**整条通道沉默**。
426
+ * @returns {{coverage: number, matched: number, supported: number, terms: number}}
427
+ */
428
+ export function idfCoverage(query, itemText, corpusTexts) {
429
+ const q = [...new Set(tokenize(query))]
430
+ if (!q.length) return { coverage: 0, matched: 0, supported: 0, terms: 0 }
431
+ const item = new Set(tokenize(itemText))
432
+ const corpus = (corpusTexts || []).map((t) => new Set(tokenize(t)))
433
+ const n = Math.max(corpus.length, 1)
434
+ let total = 0
435
+ let hit = 0
436
+ let matched = 0
437
+ let supported = 0
438
+ for (const term of q) {
439
+ let df = 0
440
+ for (const doc of corpus) if (doc.has(term)) df++
441
+ // df=0 的词不计入分母:它对"选哪一条"没有分辨力。但 supported 会记下有多少词是
442
+ // 语料根本无法表示的——那是"这条查询整体上离语料太远"的证据,交给调用方决定沉默。
443
+ if (df === 0) continue
444
+ supported++
445
+ const w = Math.log(1 + n / (1 + df))
446
+ total += w
447
+ if (item.has(term)) {
448
+ hit += w
449
+ matched++
450
+ }
451
+ }
452
+ return { coverage: total > 0 ? hit / total : 0, matched, supported, terms: q.length }
453
+ }
454
+
455
+ /**
456
+ * 层内相对阈值:命中分至少达到该层最高分的 `ratioMin`。尺度无关(BM25 分数无上界),
457
+ * 取代"整条消息 vs 整条 insight 的集合重叠 ÷ 较大集合"这种长度敏感判据。
458
+ * @param {{ id?: string, score: number }[]} scored - 已按分数降序的候选。
459
+ * @param {{ ratioMin?: number }} [opts] - 相对阈值,默认 0.5。
460
+ * @returns {object[]} 通过阈值的候选(零分恒被剔除)。
461
+ */
462
+ export function relativeHits(scored, { ratioMin = 0.5 } = {}) {
463
+ const list = (scored || []).filter((r) => r && Number.isFinite(r.score) && r.score > 0)
464
+ if (!list.length) return []
465
+ const top = list[0].score
466
+ const floor = top * (typeof ratioMin === 'number' && ratioMin > 0 ? ratioMin : 0.5)
467
+ return list.filter((r) => r.score >= floor)
468
+ }
package/src/recall.js ADDED
@@ -0,0 +1,217 @@
1
+ /**
2
+ * 统一召回核心(pull 侧)。
3
+ *
4
+ * 设计口径:「一个记忆,两种交付契约」——recall 回答"记忆里有什么和这件事有关",
5
+ * readiness 回答"动手之前必须知道什么"(PR2)。本模块只做前者,且是 pull/push 共用
6
+ * 的检索核心,避免第二次为 insight 层另写一个更弱的匹配器。
7
+ *
8
+ * 三条不变量:
9
+ * 1. 每层一个适配器:把该层条目映射到同一个检索条目契约
10
+ * ({ title, keywords, summary, terms, sourcePath },见 util/search.js:weightedFieldText)。
11
+ * 2. 每层各自 top-k 后再合并:20k doc 条目不能把十几条 insight 挤出结果。
12
+ * 3. 跨层不给"伪可比"的单一分数:桶内先按层内 top 归一为 rel,再乘层先验;
13
+ * 该分数只用于提示级排序,pull 的展示始终按层分节。
14
+ *
15
+ * 规范段提权:标题命中"必须遵守 / 禁止 / rules / must"等词集的 doc chunk 在召回中提权。
16
+ * 确定性、零模型、零依赖——不违反「零依赖、零后台、零 LLM 索引期」。
17
+ *
18
+ * @module dsh-project-memory/recall
19
+ */
20
+ import { rankEntriesMergedScored, rankEntriesStreaming, rankExperienceScored } from './util/search.js'
21
+
22
+ /** 各层先验:insight 是"踩过的坑",略高于普通文档;经验层已由 experience 文件服务,略低。 */
23
+ export const LAYER_PRIORS = Object.freeze({
24
+ insight: 1.15,
25
+ doc: 1,
26
+ symbol: 1,
27
+ experience: 0.9,
28
+ })
29
+
30
+ /** 规范段词集(多语言、确定性):标题命中即视为"规约/清单"段。 */
31
+ export const NORMATIVE_HEADING =
32
+ /(必须|禁止|不得|不要|铁律|规约|规范|规则|约定|清单|检查表|checklist|rules?|requirements?|must|must not|do not|don't|never)/i
33
+
34
+ /** 规范段提权倍数。取 1.5 与 CJK 短语加成同量级:足以翻转词频差,不足以压过强命中。 */
35
+ export const NORMATIVE_PRIOR = 1.5
36
+
37
+ /**
38
+ * 规范段先验:doc 条目标题命中规范词集时提权,其余条目恒为 1。
39
+ * @param {{ title?: string }} entry - 任意检索条目(通常是 doc chunk)。
40
+ * @returns {number} 1 或 {@link NORMATIVE_PRIOR}。
41
+ */
42
+ export function normativePrior(entry) {
43
+ return NORMATIVE_HEADING.test(String(entry?.title || '')) ? NORMATIVE_PRIOR : 1
44
+ }
45
+
46
+ /**
47
+ * 一条 insight 的正文文本(用于展示、注入与检索兜底)。
48
+ * 顺序即优先级:fix > choice > solution > pattern/problem > body > steps。
49
+ * @param {object} ins - 归一化后的 insight 条目。
50
+ * @returns {string} 单行正文。
51
+ */
52
+ export function insightBodyText(ins) {
53
+ const parts = []
54
+ if (ins?.fix) parts.push(ins.fix)
55
+ else if (ins?.choice) parts.push(ins.choice)
56
+ else if (ins?.solution) parts.push(ins.solution)
57
+ else if (ins?.pattern) parts.push(ins.pattern)
58
+ else if (ins?.problem) parts.push(ins.problem)
59
+ else if (ins?.body) parts.push(ins.body)
60
+ if (ins?.reason) parts.push(`理由:${ins.reason}`)
61
+ if (Array.isArray(ins?.steps) && ins.steps.length) {
62
+ parts.push(ins.steps.map((s, i) => `${i + 1}. ${s}`).join(' '))
63
+ }
64
+ return parts.filter(Boolean).join(' ')
65
+ }
66
+
67
+ /**
68
+ * insight → 统一检索条目契约的适配器。
69
+ * 复用 doc 条目的字段语义,于是 title×5 字段加权、CJK 短语加成、IDF 打分全部现成可用。
70
+ * @param {object} ins - 归一化后的 insight 条目。
71
+ * @returns {object} 检索条目(额外字段仅供渲染,不参与打分)。
72
+ */
73
+ export function insightToEntry(ins) {
74
+ const derived = ins?.triggerDerived || {}
75
+ const keywords = [
76
+ ins?.kind,
77
+ ins?.scope,
78
+ ...(Array.isArray(ins?.files) ? ins.files : []),
79
+ ...(Array.isArray(ins?.symbols) ? ins.symbols : []),
80
+ ...(Array.isArray(ins?.tags) ? ins.tags : []),
81
+ // 派生信号(PR3):只提升召回,不参与强制注入
82
+ ...(Array.isArray(derived.keywords) ? derived.keywords : []),
83
+ ...(Array.isArray(derived.actions) ? derived.actions : []),
84
+ ].filter(Boolean)
85
+ return {
86
+ id: `insight:${ins?.id}`,
87
+ insightId: ins?.id,
88
+ type: 'insight',
89
+ kind: ins?.kind,
90
+ scope: ins?.scope,
91
+ confidence: ins?.confidence,
92
+ files: Array.isArray(ins?.files) ? ins.files : [],
93
+ symbols: Array.isArray(ins?.symbols) ? ins.symbols : [],
94
+ title: String(ins?.title || '(untitled)'),
95
+ summary: insightBodyText(ins).slice(0, 300),
96
+ // 检索用词项:pattern/problem/reason 是"症状"侧词汇,fix 已在 summary 里(title×5 之外单算 1 份)。
97
+ terms: [ins?.pattern, ins?.problem, ins?.reason, ...(Array.isArray(derived.paths) ? derived.paths : [])].filter(Boolean).join(' '),
98
+ keywords,
99
+ sourcePath: Array.isArray(ins?.files) && ins.files.length ? ins.files[0] : '',
100
+ status: 'exact',
101
+ }
102
+ }
103
+
104
+ /**
105
+ * 召回可见的 insight 集合:作用域可见性跟随会话绑定。
106
+ * - project / global:始终可见(归档除外);
107
+ * - task:只在该会话绑定了对应任务时可见(任务私有,不泄露给别的会话);
108
+ * - 迁移影子(kind=experience 且 source=migrate)不在此列:它们由 experience 层服务,
109
+ * 否则同一内容会在 `## Insights` 与 `## Experience` 里各出现一次。
110
+ * @param {{ store?: object, globalStore?: object, boundTaskId?: string|null }} opts
111
+ * @returns {object[]} 可检索的 insight 条目。
112
+ */
113
+ export function visibleInsights({ store = null, globalStore = null, boundTaskId = null } = {}) {
114
+ const out = []
115
+ for (const it of (store && typeof store.insightItems === 'function' ? store.insightItems() : []) || []) {
116
+ if (!it || it.archived) continue
117
+ if (it.scope === 'task') continue
118
+ if (it.draft === true) continue
119
+ if (it.kind === 'experience' && it.source === 'migrate') continue // v0.4 影子,交给 experience 层
120
+ out.push(it)
121
+ }
122
+ for (const it of (globalStore && typeof globalStore.items === 'function' ? globalStore.items() : []) || []) {
123
+ if (!it || it.archived || it.draft === true) continue
124
+ out.push(it)
125
+ }
126
+ if (boundTaskId && store && typeof store.getTask === 'function') {
127
+ const task = store.getTask(boundTaskId)
128
+ for (const it of (task && task.insights) || []) {
129
+ if (!it || it.archived) continue
130
+ out.push({ ...it, scope: it.scope || 'task' })
131
+ }
132
+ }
133
+ return out
134
+ }
135
+
136
+ /**
137
+ * 按层召回:每层各自 top-k,桶内保留原始分与层内相对分。
138
+ * @param {object} opts
139
+ * @param {object} [opts.store] - ProjectMemoryStore;提供 doc/symbol/experience 池与 IDF 缓存。
140
+ * @param {object} [opts.globalStore] - GlobalStore;提供 global 级 insight。
141
+ * @param {object[]} [opts.entries] - 直接给定检索池(测试/自定义场景),优先于 store。
142
+ * @param {string[]|string} [opts.queries] - 一条或多条查询(LLM 扩展变体也走这里)。
143
+ * @param {string[]} [opts.layers] - 要召回的层;默认四层。
144
+ * @param {number} [opts.limit] - 每层上限。
145
+ * @param {string|null} [opts.boundTaskId] - 当前会话绑定的任务(决定 task 级可见性)。
146
+ * @returns {{ layers: object[], flat: object[] }} layers 为分桶结果,flat 为归一化+先验后的跨层提示序。
147
+ */
148
+ export function recallItems(opts = {}) {
149
+ const {
150
+ store = null,
151
+ globalStore = null,
152
+ entries = null,
153
+ queries = [],
154
+ layers = ['doc', 'symbol', 'experience', 'insight'],
155
+ limit = 8,
156
+ boundTaskId = null,
157
+ } = opts
158
+ const qs = (Array.isArray(queries) ? queries : [queries]).map((q) => String(q ?? '')).filter(Boolean)
159
+ const want = new Set(layers)
160
+ const idf = store && typeof store.getIdfCache === 'function' ? store.getIdfCache() : {}
161
+ const buckets = []
162
+
163
+ if (want.has('doc') || want.has('symbol')) {
164
+ const pool = (entries || (store && typeof store.allEntries === 'function' ? store.allEntries() : []) || [])
165
+ .filter((e) => e && (e.type === 'doc' || e.type === 'symbol') && want.has(e.type))
166
+ const scored = rankEntriesStreaming(pool, qs, idf, limit)
167
+ const byLayer = new Map([['doc', []], ['symbol', []]])
168
+ for (const { entry, score } of scored) {
169
+ const prior = entry.type === 'doc' ? normativePrior(entry) : LAYER_PRIORS.symbol
170
+ byLayer.get(entry.type).push({ item: entry, score, prior, weightedScore: score * prior })
171
+ }
172
+ for (const layer of ['doc', 'symbol']) {
173
+ const hits = byLayer.get(layer)
174
+ if (!hits.length) continue
175
+ hits.sort((a, b) => b.weightedScore - a.weightedScore)
176
+ const top = hits[0].weightedScore || 1
177
+ buckets.push({
178
+ layer,
179
+ prior: LAYER_PRIORS[layer],
180
+ hits: hits.map((h) => ({ ...h, rel: h.weightedScore / top })),
181
+ })
182
+ }
183
+ }
184
+
185
+ if (want.has('experience')) {
186
+ const scored = rankExperienceScored((store && store.experience) || [], qs, limit)
187
+ if (scored.length) {
188
+ const top = scored[0].score || 1
189
+ buckets.push({
190
+ layer: 'experience',
191
+ prior: LAYER_PRIORS.experience,
192
+ hits: scored.map(({ item, score }) => ({ item, score, prior: LAYER_PRIORS.experience, weightedScore: score, rel: score / top })),
193
+ })
194
+ }
195
+ }
196
+
197
+ if (want.has('insight')) {
198
+ const pool = entries !== null
199
+ ? entries.filter((e) => e && e.type === 'insight')
200
+ : visibleInsights({ store, globalStore, boundTaskId }).map(insightToEntry)
201
+ const scored = rankEntriesMergedScored(pool, qs, limit)
202
+ if (scored.length) {
203
+ const top = scored[0].score || 1
204
+ buckets.push({
205
+ layer: 'insight',
206
+ prior: LAYER_PRIORS.insight,
207
+ hits: scored.map(({ entry, score }) => ({ item: entry, score, prior: LAYER_PRIORS.insight, weightedScore: score, rel: score / top })),
208
+ })
209
+ }
210
+ }
211
+
212
+ const flat = buckets
213
+ .flatMap((b) => b.hits.map((h) => ({ ...h, layer: b.layer, layerPrior: b.prior, hint: h.rel * b.prior })))
214
+ .sort((a, b) => b.hint - a.hint)
215
+ .slice(0, limit)
216
+ return { layers: buckets, flat }
217
+ }