@yolk_vat-y/dsh-project-memory 0.5.4 → 0.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +105 -0
- package/README.md +75 -42
- package/README.zh-CN.md +75 -42
- package/client/client.js +111 -92
- package/client/client.js.map +1 -1
- package/package.json +5 -2
- package/scripts/bench-store.mjs +113 -0
- package/scripts/bench-synthetic.mjs +364 -0
- package/scripts/bench.mjs +396 -0
- package/src/auto-inject.js +246 -51
- package/src/client/MemoryView.tsx +4 -2
- package/src/client/TaskComponents.tsx +7 -6
- package/src/client/TaskPanel.module.css +7 -0
- package/src/client/TaskPanel.tsx +12 -8
- package/src/client/task-ui-store.ts +11 -1
- package/src/index.js +19 -2
- package/src/insight-store.js +6 -4
- package/src/parsers/pdfjs-parser.js +3 -0
- package/src/readiness.js +224 -0
- package/src/recall.js +217 -0
- package/src/store.js +20 -1
- package/src/tools/index-repo.js +7 -1
- package/src/tools/lesson-tools.js +7 -2
- package/src/tools/query-memory.js +61 -23
- package/src/tools/watch-repo.js +10 -1
- package/src/util/fs.js +41 -0
- package/src/watch.js +28 -4
package/src/insight-store.js
CHANGED
|
@@ -98,10 +98,12 @@ export function normalizeInsight(raw, extra = {}) {
|
|
|
98
98
|
if (Array.isArray(raw[f]) && raw[f].length) ins[f] = [...new Set(raw[f].map((x) => String(x)))]
|
|
99
99
|
}
|
|
100
100
|
if (raw.trigger && typeof raw.trigger === 'object') {
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
101
|
+
// 所有 kind 通用:keywords / symbols / actions / paths / scope(actions/paths 见 src/readiness.js)
|
|
102
|
+
const tr = {}
|
|
103
|
+
for (const f of ['keywords', 'symbols', 'actions', 'paths', 'scope']) {
|
|
104
|
+
if (Array.isArray(raw.trigger[f]) && raw.trigger[f].length) tr[f] = raw.trigger[f].map(String)
|
|
105
|
+
}
|
|
106
|
+
if (Object.keys(tr).length) ins.trigger = tr
|
|
105
107
|
}
|
|
106
108
|
return ins
|
|
107
109
|
}
|
|
@@ -25,6 +25,9 @@ const PDFJS_OPTIONS = {
|
|
|
25
25
|
isEvalSupported: false,
|
|
26
26
|
useWorkerFetch: false,
|
|
27
27
|
useWorker: false,
|
|
28
|
+
// 0 = ERRORS。损坏/线性化缺失的 PDF 会让 pdf.js 打 "Warning: Indexing all PDF objects"
|
|
29
|
+
// 之类的告警——那是它自己的恢复路径,对使用者没有可操作性,静音。
|
|
30
|
+
verbosity: 0,
|
|
28
31
|
}
|
|
29
32
|
|
|
30
33
|
function buildMarkdown(pages) {
|
package/src/readiness.js
ADDED
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 就绪引擎(push 侧)。
|
|
3
|
+
*
|
|
4
|
+
* recall 回答"记忆里有什么和这件事有关";readiness 回答"动手之前必须知道什么"。
|
|
5
|
+
* 交付契约(设计核心):
|
|
6
|
+
* - **authored trigger = 确定性注入**:作者写下的 keywords / symbols / actions / paths 命中即在
|
|
7
|
+
* 动手前全文注入,优先级 1。可审计(返回 `why`),不参与任何相似度打分。
|
|
8
|
+
* - **统计信号 = 提示态**:没有 trigger 的条目只能作为截断提示进入,优先级 2,且判据是
|
|
9
|
+
* **层内相对阈值**(相对最高分),不是长度敏感的魔法常量。
|
|
10
|
+
*
|
|
11
|
+
* 两个窗口(忠于宿主契约:DSH 没有"工具执行前拦截"钩子,唯一能加上下文的缝是 agent/pre-step):
|
|
12
|
+
* - 先发窗口:人类消息里的意图词("提交/发布/公开")→ 动作 id 在模型规划动作之前就命中;
|
|
13
|
+
* - 反应窗口:已观察到的 tool/call 参数(`git commit`、`npm publish`、改动的路径)走同一个匹配器。
|
|
14
|
+
*
|
|
15
|
+
* 纯函数、零依赖、不调用模型。
|
|
16
|
+
* @module dsh-project-memory/readiness
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* 动作词典:动作 id → 匹配模式。命令形态与人类意图词都归一成同一套 id,
|
|
21
|
+
* 于是"你顺便提交一下"与 `git commit -m x` 对 trigger 是同一件事。
|
|
22
|
+
*/
|
|
23
|
+
export const ACTION_LEXICON = [
|
|
24
|
+
['git-add', [/git\s+add\b/i]],
|
|
25
|
+
['git-commit', [/git\s+commit\b/i, /提交/]],
|
|
26
|
+
['git-push', [/git\s+push\b/i, /推送/]],
|
|
27
|
+
['npm-publish', [/npm\s+publish\b/i, /发包/, /发布\s*到\s*npm/i]],
|
|
28
|
+
['release', [/git\s+tag\b/i, /\brelease\b/i, /发版/, /发布版本/]],
|
|
29
|
+
['go-public', [/公开/, /开源/, /对外/, /上线/]],
|
|
30
|
+
['delete', [/rm\s+-rf\b/, /git\s+rm\b/, /删除/, /清理/]],
|
|
31
|
+
['migrate', [/migrate\b/i, /迁移/]],
|
|
32
|
+
['deploy', [/\bdeploy\b/i, /部署/]],
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
/** 带扩展名的路径 token(README.md、src/util/fs.js …)。 */
|
|
36
|
+
const PATH_TOKEN = /(?:[A-Za-z0-9_.@-]+\/)*[A-Za-z0-9_.@-]+\.[A-Za-z][A-Za-z0-9]{0,7}\b/g
|
|
37
|
+
/** 点开头的裸文件名(.gitignore、.npmrc …)。 */
|
|
38
|
+
const DOTFILE_TOKEN = /(?:^|[\s"'`(])(\.[A-Za-z0-9_-]{2,})/g
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* 文本 → 动作 id 列表(确定性、去重、词典序稳定)。
|
|
42
|
+
* @param {string} text - 人类消息或工具调用参数。
|
|
43
|
+
* @returns {string[]} 动作 id。
|
|
44
|
+
*/
|
|
45
|
+
export function detectActions(text) {
|
|
46
|
+
const s = String(text || '')
|
|
47
|
+
if (!s) return []
|
|
48
|
+
const out = []
|
|
49
|
+
for (const [id, patterns] of ACTION_LEXICON) {
|
|
50
|
+
if (patterns.some((re) => re.test(s))) out.push(id)
|
|
51
|
+
}
|
|
52
|
+
return out
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* 文本 → 路径 token(用于 trigger.paths 的 glob 匹配)。
|
|
57
|
+
* @param {string} text - 工具调用参数或人类消息。
|
|
58
|
+
* @returns {string[]} 去重后的路径 token。
|
|
59
|
+
*/
|
|
60
|
+
export function extractPaths(text) {
|
|
61
|
+
const s = String(text || '')
|
|
62
|
+
if (!s) return []
|
|
63
|
+
const out = new Set()
|
|
64
|
+
for (const m of s.matchAll(PATH_TOKEN)) out.add(m[0])
|
|
65
|
+
for (const m of s.matchAll(DOTFILE_TOKEN)) out.add(m[1])
|
|
66
|
+
return [...out]
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* 组装就绪上下文:人类意图 + 已观察动作。三处来源都做动作/路径归一,调用方只需给文本。
|
|
71
|
+
* @param {object} input
|
|
72
|
+
* @param {string} [input.humanText] - 本轮人类消息(先发窗口)。
|
|
73
|
+
* @param {string} [input.actionText] - 已观察到的 tool/call 参数(反应窗口)。
|
|
74
|
+
* @param {string[]} [input.actions] - 调用方已归一化的动作 id(可选,会被并集)。
|
|
75
|
+
* @param {string[]} [input.paths] - 调用方已归一化的路径(可选,会被并集)。
|
|
76
|
+
* @returns {{ humanText: string, actionText: string, actions: string[], paths: string[] }}
|
|
77
|
+
*/
|
|
78
|
+
export function buildReadinessContext(input = {}) {
|
|
79
|
+
const humanText = String(input.humanText ?? input.query ?? '')
|
|
80
|
+
const actionText = String(input.actionText || '')
|
|
81
|
+
const actions = new Set([...(input.actions || []), ...detectActions(humanText), ...detectActions(actionText)])
|
|
82
|
+
const paths = new Set([...(input.paths || []), ...extractPaths(actionText), ...extractPaths(humanText)])
|
|
83
|
+
return { humanText, actionText, actions: [...actions], paths: [...paths] }
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* 派生 trigger 的意图词表(有界、确定性)。派生信号**只进提示通道**:它能提升召回,
|
|
88
|
+
* 但永不强制注入——"自动学出来的东西"不该污染上下文。
|
|
89
|
+
*/
|
|
90
|
+
export const ACTION_WORDS = [
|
|
91
|
+
'提交', '公开', '泄漏', '发布', '发包', '发版', '上线', '部署', '迁移', '删除', '清理', '回滚',
|
|
92
|
+
'推送', '打包', '构建', '依赖', '密钥', '权限', '并发', '超时', '内存', '性能', '安全', '面试',
|
|
93
|
+
]
|
|
94
|
+
|
|
95
|
+
/** 无扩展名但明确是文件名的裸词(README / CHANGELOG / …),当关键词用。 */
|
|
96
|
+
const BARE_FILE_NAMES = /\b(?:README|CHANGELOG|LICENSE|AGENTS|CONTRIBUTING|Dockerfile|Makefile)\b/g
|
|
97
|
+
|
|
98
|
+
const DERIVED_MAX_KEYWORDS = 6
|
|
99
|
+
const DERIVED_MAX_ACTIONS = 4
|
|
100
|
+
const DERIVED_MAX_PATHS = 6
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* 从一条 insight 的正文确定性派生触发信号(零模型、可重放)。
|
|
104
|
+
* 结果只写进 `triggerDerived`,供提示通道(检索文本)使用;是否强制注入只由 authored `trigger` 决定。
|
|
105
|
+
* @param {object} ins - 归一化后的 insight 条目。
|
|
106
|
+
* @returns {{ keywords: string[], actions: string[], paths: string[] }}
|
|
107
|
+
*/
|
|
108
|
+
export function deriveTrigger(ins) {
|
|
109
|
+
const text = [
|
|
110
|
+
ins?.title,
|
|
111
|
+
ins?.pattern,
|
|
112
|
+
ins?.fix,
|
|
113
|
+
ins?.choice,
|
|
114
|
+
ins?.reason,
|
|
115
|
+
ins?.problem,
|
|
116
|
+
ins?.solution,
|
|
117
|
+
ins?.topic,
|
|
118
|
+
ins?.body,
|
|
119
|
+
...(Array.isArray(ins?.steps) ? ins.steps : []),
|
|
120
|
+
].filter(Boolean).join('\n')
|
|
121
|
+
const bare = text.match(BARE_FILE_NAMES) || []
|
|
122
|
+
const keywords = [...new Set([...ACTION_WORDS.filter((w) => text.includes(w)), ...bare])].slice(0, DERIVED_MAX_KEYWORDS)
|
|
123
|
+
const actions = detectActions(text).slice(0, DERIVED_MAX_ACTIONS)
|
|
124
|
+
const paths = extractPaths(text).slice(0, DERIVED_MAX_PATHS)
|
|
125
|
+
return { keywords, actions, paths }
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* insights 文档的懒回填:给缺少 `triggerDerived` 的条目补上派生信号(增量、幂等)。
|
|
130
|
+
* 只改内存;是否落盘由调用方的 commit 决定。
|
|
131
|
+
* @param {{ items?: object[] }} doc - insights 文档。
|
|
132
|
+
* @returns {boolean} 是否发生了变更(用于置 dirty)。
|
|
133
|
+
*/
|
|
134
|
+
export function backfillDerivedTriggers(doc) {
|
|
135
|
+
if (!doc || !Array.isArray(doc.items)) return false
|
|
136
|
+
let changed = false
|
|
137
|
+
for (const it of doc.items) {
|
|
138
|
+
if (!it || it.triggerDerived) continue
|
|
139
|
+
const derived = deriveTrigger(it)
|
|
140
|
+
if (!derived.keywords.length && !derived.actions.length && !derived.paths.length) continue
|
|
141
|
+
it.triggerDerived = derived
|
|
142
|
+
changed = true
|
|
143
|
+
}
|
|
144
|
+
return changed
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* 提示通道的查询文本:剔除 1–2 个字符的拉丁 token。
|
|
149
|
+
*
|
|
150
|
+
* 为什么:`PR` / `CI` / `OS` 这类缩写太短、歧义太大,一个巧合命中就能当上该层最高分,
|
|
151
|
+
* 于是以 `relative:1.00` 混进上下文(实测:人类消息里的 "PR" 把一条 task-tools 越权 lesson
|
|
152
|
+
* 顶到了提示位)。内容词(≥3 字符的拉丁标识符、CJK 词)不受影响;
|
|
153
|
+
* authored trigger 通道完全不走这里,确定性匹配保持字面语义。
|
|
154
|
+
* @param {string} text - 人类消息或工具参数。
|
|
155
|
+
* @returns {string} 过滤后的查询文本。
|
|
156
|
+
*/
|
|
157
|
+
export function hintQueryText(text) {
|
|
158
|
+
return String(text || '')
|
|
159
|
+
.split(/\s+/)
|
|
160
|
+
// 路径形态的 token 原样保留(src/util/fs.js 里的 fs / js 是有效证据),
|
|
161
|
+
// 只对独立词做缩写剔除。
|
|
162
|
+
.map((w) => (/[/.]/.test(w) ? w : w.replace(/\b[A-Za-z0-9_]{1,2}\b/g, ' ')))
|
|
163
|
+
.join(' ')
|
|
164
|
+
.replace(/\s+/g, ' ')
|
|
165
|
+
.trim()
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** glob(只支持 `*`)→ 正则。 */
|
|
169
|
+
function globToRegExp(pattern) {
|
|
170
|
+
const escaped = String(pattern).replace(/[.+^${}()|[\]\\]/g, '\\$&').replace(/\*/g, '.*')
|
|
171
|
+
return new RegExp(`^${escaped}$`, 'i')
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* trigger 命中判定(确定性,无打分)。
|
|
176
|
+
* @param {object|null} trigger - `{ keywords?, symbols?, actions?, paths?, scope? }`。
|
|
177
|
+
* @param {object} ctx - {@link buildReadinessContext} 的结果。
|
|
178
|
+
* @returns {string|null} 命中原因(`keyword:…` / `symbol:…` / `action:…` / `path:…`),未命中为 null。
|
|
179
|
+
*/
|
|
180
|
+
export function matchTrigger(trigger, ctx) {
|
|
181
|
+
if (!trigger) return null
|
|
182
|
+
const haystack = `${ctx?.humanText || ''}\n${ctx?.actionText || ''}`
|
|
183
|
+
const lower = haystack.toLowerCase()
|
|
184
|
+
for (const kw of trigger.keywords || []) {
|
|
185
|
+
const k = String(kw || '')
|
|
186
|
+
if (k && lower.includes(k.toLowerCase())) return `keyword:${k}`
|
|
187
|
+
}
|
|
188
|
+
for (const sym of trigger.symbols || []) {
|
|
189
|
+
const s = String(sym || '')
|
|
190
|
+
if (s && haystack.includes(s)) return `symbol:${s}`
|
|
191
|
+
}
|
|
192
|
+
const actions = new Set(ctx?.actions || [])
|
|
193
|
+
for (const a of trigger.actions || []) {
|
|
194
|
+
const id = String(a || '')
|
|
195
|
+
if (id && actions.has(id)) return `action:${id}`
|
|
196
|
+
}
|
|
197
|
+
for (const p of trigger.paths || []) {
|
|
198
|
+
const pattern = String(p || '')
|
|
199
|
+
if (!pattern) continue
|
|
200
|
+
const re = globToRegExp(pattern)
|
|
201
|
+
const hit = (ctx?.paths || []).some((cand) => {
|
|
202
|
+
const c = String(cand || '')
|
|
203
|
+
if (!c) return false
|
|
204
|
+
return re.test(c) || re.test(c.split(/[\\/]/).pop() || '')
|
|
205
|
+
})
|
|
206
|
+
if (hit) return `path:${pattern}`
|
|
207
|
+
}
|
|
208
|
+
return null
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* 层内相对阈值:命中分至少达到该层最高分的 `ratioMin`。尺度无关(BM25 分数无上界),
|
|
213
|
+
* 取代"整条消息 vs 整条 insight 的集合重叠 ÷ 较大集合"这种长度敏感判据。
|
|
214
|
+
* @param {{ id?: string, score: number }[]} scored - 已按分数降序的候选。
|
|
215
|
+
* @param {{ ratioMin?: number }} [opts] - 相对阈值,默认 0.5。
|
|
216
|
+
* @returns {object[]} 通过阈值的候选(零分恒被剔除)。
|
|
217
|
+
*/
|
|
218
|
+
export function relativeHits(scored, { ratioMin = 0.5 } = {}) {
|
|
219
|
+
const list = (scored || []).filter((r) => r && Number.isFinite(r.score) && r.score > 0)
|
|
220
|
+
if (!list.length) return []
|
|
221
|
+
const top = list[0].score
|
|
222
|
+
const floor = top * (typeof ratioMin === 'number' && ratioMin > 0 ? ratioMin : 0.5)
|
|
223
|
+
return list.filter((r) => r.score >= floor)
|
|
224
|
+
}
|
package/src/recall.js
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 统一召回核心(pull 侧)。
|
|
3
|
+
*
|
|
4
|
+
* 设计口径:「一个记忆,两种交付契约」——recall 回答"记忆里有什么和这件事有关",
|
|
5
|
+
* readiness 回答"动手之前必须知道什么"(PR2)。本模块只做前者,且是 pull/push 共用
|
|
6
|
+
* 的检索核心,避免第二次为 insight 层另写一个更弱的匹配器。
|
|
7
|
+
*
|
|
8
|
+
* 三条不变量:
|
|
9
|
+
* 1. 每层一个适配器:把该层条目映射到同一个检索条目契约
|
|
10
|
+
* ({ title, keywords, summary, terms, sourcePath },见 util/search.js:weightedFieldText)。
|
|
11
|
+
* 2. 每层各自 top-k 后再合并:20k doc 条目不能把十几条 insight 挤出结果。
|
|
12
|
+
* 3. 跨层不给"伪可比"的单一分数:桶内先按层内 top 归一为 rel,再乘层先验;
|
|
13
|
+
* 该分数只用于提示级排序,pull 的展示始终按层分节。
|
|
14
|
+
*
|
|
15
|
+
* 规范段提权:标题命中"必须遵守 / 禁止 / rules / must"等词集的 doc chunk 在召回中提权。
|
|
16
|
+
* 确定性、零模型、零依赖——不违反「零依赖、零后台、零 LLM 索引期」。
|
|
17
|
+
*
|
|
18
|
+
* @module dsh-project-memory/recall
|
|
19
|
+
*/
|
|
20
|
+
import { rankEntriesMergedScored, rankEntriesStreaming, rankExperienceScored } from './util/search.js'
|
|
21
|
+
|
|
22
|
+
/** 各层先验:insight 是"踩过的坑",略高于普通文档;经验层已由 experience 文件服务,略低。 */
|
|
23
|
+
export const LAYER_PRIORS = Object.freeze({
|
|
24
|
+
insight: 1.15,
|
|
25
|
+
doc: 1,
|
|
26
|
+
symbol: 1,
|
|
27
|
+
experience: 0.9,
|
|
28
|
+
})
|
|
29
|
+
|
|
30
|
+
/** 规范段词集(多语言、确定性):标题命中即视为"规约/清单"段。 */
|
|
31
|
+
export const NORMATIVE_HEADING =
|
|
32
|
+
/(必须|禁止|不得|不要|铁律|规约|规范|规则|约定|清单|检查表|checklist|rules?|requirements?|must|must not|do not|don't|never)/i
|
|
33
|
+
|
|
34
|
+
/** 规范段提权倍数。取 1.5 与 CJK 短语加成同量级:足以翻转词频差,不足以压过强命中。 */
|
|
35
|
+
export const NORMATIVE_PRIOR = 1.5
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* 规范段先验:doc 条目标题命中规范词集时提权,其余条目恒为 1。
|
|
39
|
+
* @param {{ title?: string }} entry - 任意检索条目(通常是 doc chunk)。
|
|
40
|
+
* @returns {number} 1 或 {@link NORMATIVE_PRIOR}。
|
|
41
|
+
*/
|
|
42
|
+
export function normativePrior(entry) {
|
|
43
|
+
return NORMATIVE_HEADING.test(String(entry?.title || '')) ? NORMATIVE_PRIOR : 1
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* 一条 insight 的正文文本(用于展示、注入与检索兜底)。
|
|
48
|
+
* 顺序即优先级:fix > choice > solution > pattern/problem > body > steps。
|
|
49
|
+
* @param {object} ins - 归一化后的 insight 条目。
|
|
50
|
+
* @returns {string} 单行正文。
|
|
51
|
+
*/
|
|
52
|
+
export function insightBodyText(ins) {
|
|
53
|
+
const parts = []
|
|
54
|
+
if (ins?.fix) parts.push(ins.fix)
|
|
55
|
+
else if (ins?.choice) parts.push(ins.choice)
|
|
56
|
+
else if (ins?.solution) parts.push(ins.solution)
|
|
57
|
+
else if (ins?.pattern) parts.push(ins.pattern)
|
|
58
|
+
else if (ins?.problem) parts.push(ins.problem)
|
|
59
|
+
else if (ins?.body) parts.push(ins.body)
|
|
60
|
+
if (ins?.reason) parts.push(`理由:${ins.reason}`)
|
|
61
|
+
if (Array.isArray(ins?.steps) && ins.steps.length) {
|
|
62
|
+
parts.push(ins.steps.map((s, i) => `${i + 1}. ${s}`).join(' '))
|
|
63
|
+
}
|
|
64
|
+
return parts.filter(Boolean).join(' ')
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* insight → 统一检索条目契约的适配器。
|
|
69
|
+
* 复用 doc 条目的字段语义,于是 title×5 字段加权、CJK 短语加成、IDF 打分全部现成可用。
|
|
70
|
+
* @param {object} ins - 归一化后的 insight 条目。
|
|
71
|
+
* @returns {object} 检索条目(额外字段仅供渲染,不参与打分)。
|
|
72
|
+
*/
|
|
73
|
+
export function insightToEntry(ins) {
|
|
74
|
+
const derived = ins?.triggerDerived || {}
|
|
75
|
+
const keywords = [
|
|
76
|
+
ins?.kind,
|
|
77
|
+
ins?.scope,
|
|
78
|
+
...(Array.isArray(ins?.files) ? ins.files : []),
|
|
79
|
+
...(Array.isArray(ins?.symbols) ? ins.symbols : []),
|
|
80
|
+
...(Array.isArray(ins?.tags) ? ins.tags : []),
|
|
81
|
+
// 派生信号(PR3):只提升召回,不参与强制注入
|
|
82
|
+
...(Array.isArray(derived.keywords) ? derived.keywords : []),
|
|
83
|
+
...(Array.isArray(derived.actions) ? derived.actions : []),
|
|
84
|
+
].filter(Boolean)
|
|
85
|
+
return {
|
|
86
|
+
id: `insight:${ins?.id}`,
|
|
87
|
+
insightId: ins?.id,
|
|
88
|
+
type: 'insight',
|
|
89
|
+
kind: ins?.kind,
|
|
90
|
+
scope: ins?.scope,
|
|
91
|
+
confidence: ins?.confidence,
|
|
92
|
+
files: Array.isArray(ins?.files) ? ins.files : [],
|
|
93
|
+
symbols: Array.isArray(ins?.symbols) ? ins.symbols : [],
|
|
94
|
+
title: String(ins?.title || '(untitled)'),
|
|
95
|
+
summary: insightBodyText(ins).slice(0, 300),
|
|
96
|
+
// 检索用词项:pattern/problem/reason 是"症状"侧词汇,fix 已在 summary 里(title×5 之外单算 1 份)。
|
|
97
|
+
terms: [ins?.pattern, ins?.problem, ins?.reason, ...(Array.isArray(derived.paths) ? derived.paths : [])].filter(Boolean).join(' '),
|
|
98
|
+
keywords,
|
|
99
|
+
sourcePath: Array.isArray(ins?.files) && ins.files.length ? ins.files[0] : '',
|
|
100
|
+
status: 'exact',
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* 召回可见的 insight 集合:作用域可见性跟随会话绑定。
|
|
106
|
+
* - project / global:始终可见(归档除外);
|
|
107
|
+
* - task:只在该会话绑定了对应任务时可见(任务私有,不泄露给别的会话);
|
|
108
|
+
* - 迁移影子(kind=experience 且 source=migrate)不在此列:它们由 experience 层服务,
|
|
109
|
+
* 否则同一内容会在 `## Insights` 与 `## Experience` 里各出现一次。
|
|
110
|
+
* @param {{ store?: object, globalStore?: object, boundTaskId?: string|null }} opts
|
|
111
|
+
* @returns {object[]} 可检索的 insight 条目。
|
|
112
|
+
*/
|
|
113
|
+
export function visibleInsights({ store = null, globalStore = null, boundTaskId = null } = {}) {
|
|
114
|
+
const out = []
|
|
115
|
+
for (const it of (store && typeof store.insightItems === 'function' ? store.insightItems() : []) || []) {
|
|
116
|
+
if (!it || it.archived) continue
|
|
117
|
+
if (it.scope === 'task') continue
|
|
118
|
+
if (it.draft === true) continue
|
|
119
|
+
if (it.kind === 'experience' && it.source === 'migrate') continue // v0.4 影子,交给 experience 层
|
|
120
|
+
out.push(it)
|
|
121
|
+
}
|
|
122
|
+
for (const it of (globalStore && typeof globalStore.items === 'function' ? globalStore.items() : []) || []) {
|
|
123
|
+
if (!it || it.archived || it.draft === true) continue
|
|
124
|
+
out.push(it)
|
|
125
|
+
}
|
|
126
|
+
if (boundTaskId && store && typeof store.getTask === 'function') {
|
|
127
|
+
const task = store.getTask(boundTaskId)
|
|
128
|
+
for (const it of (task && task.insights) || []) {
|
|
129
|
+
if (!it || it.archived) continue
|
|
130
|
+
out.push({ ...it, scope: it.scope || 'task' })
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
return out
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* 按层召回:每层各自 top-k,桶内保留原始分与层内相对分。
|
|
138
|
+
* @param {object} opts
|
|
139
|
+
* @param {object} [opts.store] - ProjectMemoryStore;提供 doc/symbol/experience 池与 IDF 缓存。
|
|
140
|
+
* @param {object} [opts.globalStore] - GlobalStore;提供 global 级 insight。
|
|
141
|
+
* @param {object[]} [opts.entries] - 直接给定检索池(测试/自定义场景),优先于 store。
|
|
142
|
+
* @param {string[]|string} [opts.queries] - 一条或多条查询(LLM 扩展变体也走这里)。
|
|
143
|
+
* @param {string[]} [opts.layers] - 要召回的层;默认四层。
|
|
144
|
+
* @param {number} [opts.limit] - 每层上限。
|
|
145
|
+
* @param {string|null} [opts.boundTaskId] - 当前会话绑定的任务(决定 task 级可见性)。
|
|
146
|
+
* @returns {{ layers: object[], flat: object[] }} layers 为分桶结果,flat 为归一化+先验后的跨层提示序。
|
|
147
|
+
*/
|
|
148
|
+
export function recallItems(opts = {}) {
|
|
149
|
+
const {
|
|
150
|
+
store = null,
|
|
151
|
+
globalStore = null,
|
|
152
|
+
entries = null,
|
|
153
|
+
queries = [],
|
|
154
|
+
layers = ['doc', 'symbol', 'experience', 'insight'],
|
|
155
|
+
limit = 8,
|
|
156
|
+
boundTaskId = null,
|
|
157
|
+
} = opts
|
|
158
|
+
const qs = (Array.isArray(queries) ? queries : [queries]).map((q) => String(q ?? '')).filter(Boolean)
|
|
159
|
+
const want = new Set(layers)
|
|
160
|
+
const idf = store && typeof store.getIdfCache === 'function' ? store.getIdfCache() : {}
|
|
161
|
+
const buckets = []
|
|
162
|
+
|
|
163
|
+
if (want.has('doc') || want.has('symbol')) {
|
|
164
|
+
const pool = (entries || (store && typeof store.allEntries === 'function' ? store.allEntries() : []) || [])
|
|
165
|
+
.filter((e) => e && (e.type === 'doc' || e.type === 'symbol') && want.has(e.type))
|
|
166
|
+
const scored = rankEntriesStreaming(pool, qs, idf, limit)
|
|
167
|
+
const byLayer = new Map([['doc', []], ['symbol', []]])
|
|
168
|
+
for (const { entry, score } of scored) {
|
|
169
|
+
const prior = entry.type === 'doc' ? normativePrior(entry) : LAYER_PRIORS.symbol
|
|
170
|
+
byLayer.get(entry.type).push({ item: entry, score, prior, weightedScore: score * prior })
|
|
171
|
+
}
|
|
172
|
+
for (const layer of ['doc', 'symbol']) {
|
|
173
|
+
const hits = byLayer.get(layer)
|
|
174
|
+
if (!hits.length) continue
|
|
175
|
+
hits.sort((a, b) => b.weightedScore - a.weightedScore)
|
|
176
|
+
const top = hits[0].weightedScore || 1
|
|
177
|
+
buckets.push({
|
|
178
|
+
layer,
|
|
179
|
+
prior: LAYER_PRIORS[layer],
|
|
180
|
+
hits: hits.map((h) => ({ ...h, rel: h.weightedScore / top })),
|
|
181
|
+
})
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
if (want.has('experience')) {
|
|
186
|
+
const scored = rankExperienceScored((store && store.experience) || [], qs, limit)
|
|
187
|
+
if (scored.length) {
|
|
188
|
+
const top = scored[0].score || 1
|
|
189
|
+
buckets.push({
|
|
190
|
+
layer: 'experience',
|
|
191
|
+
prior: LAYER_PRIORS.experience,
|
|
192
|
+
hits: scored.map(({ item, score }) => ({ item, score, prior: LAYER_PRIORS.experience, weightedScore: score, rel: score / top })),
|
|
193
|
+
})
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
if (want.has('insight')) {
|
|
198
|
+
const pool = entries !== null
|
|
199
|
+
? entries.filter((e) => e && e.type === 'insight')
|
|
200
|
+
: visibleInsights({ store, globalStore, boundTaskId }).map(insightToEntry)
|
|
201
|
+
const scored = rankEntriesMergedScored(pool, qs, limit)
|
|
202
|
+
if (scored.length) {
|
|
203
|
+
const top = scored[0].score || 1
|
|
204
|
+
buckets.push({
|
|
205
|
+
layer: 'insight',
|
|
206
|
+
prior: LAYER_PRIORS.insight,
|
|
207
|
+
hits: scored.map(({ entry, score }) => ({ item: entry, score, prior: LAYER_PRIORS.insight, weightedScore: score, rel: score / top })),
|
|
208
|
+
})
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
const flat = buckets
|
|
213
|
+
.flatMap((b) => b.hits.map((h) => ({ ...h, layer: b.layer, layerPrior: b.prior, hint: h.rel * b.prior })))
|
|
214
|
+
.sort((a, b) => b.hint - a.hint)
|
|
215
|
+
.slice(0, limit)
|
|
216
|
+
return { layers: buckets, flat }
|
|
217
|
+
}
|
package/src/store.js
CHANGED
|
@@ -2,6 +2,7 @@ import { createHash, randomUUID } from 'node:crypto'
|
|
|
2
2
|
import { mkdirSync, readFileSync, readdirSync, renameSync, statSync, unlinkSync, writeFileSync } from 'node:fs'
|
|
3
3
|
import path from 'node:path'
|
|
4
4
|
import { rankEntries, rankExperience, tokenize, tokenizeRaw, extractCjkPhrases, makeSearchText } from './util/search.js'
|
|
5
|
+
import { backfillDerivedTriggers } from './readiness.js'
|
|
5
6
|
|
|
6
7
|
const FORMAT_FILE = 'format.json'
|
|
7
8
|
const INDEX_FILE = 'index.json'
|
|
@@ -166,6 +167,8 @@ export class ProjectMemoryStore {
|
|
|
166
167
|
const doc = loadJson(path.join(this.dir, INSIGHTS_FILE), null)
|
|
167
168
|
this.insights = doc && typeof doc === 'object' && Array.isArray(doc.items) ? doc : { version: 1, migratedAt: null, items: [] }
|
|
168
169
|
this._migrateExperienceToInsights()
|
|
170
|
+
// PR3:v1 → v2 懒回填派生 trigger(纯确定性、幂等;不调用模型,不改写已有字段)
|
|
171
|
+
if (backfillDerivedTriggers(this.insights)) this._dirtyInsights = true
|
|
169
172
|
}
|
|
170
173
|
|
|
171
174
|
_migrateExperienceToInsights() {
|
|
@@ -244,8 +247,22 @@ export class ProjectMemoryStore {
|
|
|
244
247
|
}
|
|
245
248
|
|
|
246
249
|
save() {
|
|
247
|
-
|
|
250
|
+
// 没有脏数据就不落盘。watch 每轮对每个根都无条件 commit → save;照旧执行的话,
|
|
251
|
+
// 末尾的 `_version++` + `_idfCache = null` 会打在跨实例共享的 store 上,
|
|
252
|
+
// 等于每 15 秒清空一次 IDF 缓存,废掉查询侧的 IDF 复用(v0.3.4 的 20x)。
|
|
253
|
+
const dirty =
|
|
254
|
+
this._dirtyShards.size > 0 ||
|
|
255
|
+
this._removedShards.size > 0 ||
|
|
256
|
+
this._dirtyExperience ||
|
|
257
|
+
this._dirtyInsights ||
|
|
258
|
+
this._dirtyTasks ||
|
|
259
|
+
this._dirtyBinding ||
|
|
260
|
+
this._dirtyWatch ||
|
|
261
|
+
!this._formatWritten
|
|
262
|
+
if (dirty) mkdirSync(this.dir, { recursive: true })
|
|
263
|
+
// 崩溃遗留的 *.tmp 无论有没有脏数据都顺手清掉(两次 readdir,自带 try/catch)
|
|
248
264
|
this.cleanStaleTmp()
|
|
265
|
+
if (!dirty) return
|
|
249
266
|
if (!this._formatWritten) {
|
|
250
267
|
writeJsonAtomic(path.join(this.dir, FORMAT_FILE), { version: 2, layout: 'sharded' })
|
|
251
268
|
this._formatWritten = true
|
|
@@ -272,6 +289,8 @@ export class ProjectMemoryStore {
|
|
|
272
289
|
this._dirtyExperience = false
|
|
273
290
|
}
|
|
274
291
|
if (this._dirtyInsights) {
|
|
292
|
+
// PR3:落盘即把格式标记推到 v2(v1 → v2 是纯增量:只多一个可选的 triggerDerived)
|
|
293
|
+
if (this.insights && this.insights.version !== 2) this.insights.version = 2
|
|
275
294
|
writeJsonAtomic(path.join(this.dir, INSIGHTS_FILE), this.insights)
|
|
276
295
|
this._dirtyInsights = false
|
|
277
296
|
}
|
package/src/tools/index-repo.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { defineTool } from '@deepseek-ai/dsh-tools'
|
|
2
2
|
import path from 'node:path'
|
|
3
3
|
import { statSync } from 'node:fs'
|
|
4
|
-
import { isSupportedCode, isSupportedDoc, looksLikeDump, memoryRootFor, readFileForIndex, relativePath, storeKey, walkDir } from '../util/fs.js'
|
|
4
|
+
import { assertIndexRoot, isSupportedCode, isSupportedDoc, looksLikeDump, memoryRootFor, readFileForIndex, relativePath, storeKey, walkDir } from '../util/fs.js'
|
|
5
5
|
import { buildDocEntries } from '../doc-pipeline.js'
|
|
6
6
|
import { docEntriesNeedBackfill } from '../doc-index.js'
|
|
7
7
|
import { scanSymbols } from '../symbols.js'
|
|
@@ -10,6 +10,9 @@ import { ProjectMemoryStore } from '../store.js'
|
|
|
10
10
|
import { onFileIndexed, isTypeScriptFile } from '../enhancer.js'
|
|
11
11
|
|
|
12
12
|
export async function indexRepository(ctx, config, root, { reindex = false } = {}) {
|
|
13
|
+
// 先校验根目录:缺了它,下面 `new ProjectMemoryStore(...).load()` 的 save() 会把
|
|
14
|
+
// 不存在的 root 连同 .dsh-project-memory 一起 mkdirSync 出来。
|
|
15
|
+
assertIndexRoot(root)
|
|
13
16
|
const memoryDir = memoryRootFor(root, config.memoryDir)
|
|
14
17
|
const store = new ProjectMemoryStore(memoryDir).load()
|
|
15
18
|
|
|
@@ -147,6 +150,9 @@ export function indexRepoTool(ctx, config) {
|
|
|
147
150
|
},
|
|
148
151
|
async execute(args) {
|
|
149
152
|
const root = path.resolve(args.root)
|
|
153
|
+
// 透传原始入参:Windows 风格路径在 POSIX 上会被 resolve 成 <cwd>/D:\...,
|
|
154
|
+
// 报错时要点明这是路径风格问题,而不是“目录被删了”。
|
|
155
|
+
assertIndexRoot(root, args.root)
|
|
150
156
|
return indexRepository(ctx, config, root, { reindex: Boolean(args.reindex) })
|
|
151
157
|
},
|
|
152
158
|
})
|
|
@@ -11,7 +11,7 @@ function sessionIdOf(exec) {
|
|
|
11
11
|
}
|
|
12
12
|
|
|
13
13
|
export function lessonTool(config) {
|
|
14
|
-
const kindDesc = `insight 语义:lesson(曾踩坑/纠偏, pattern→fix) | decision(权衡选型, choice→reason) | procedure(多步指南, steps
|
|
14
|
+
const kindDesc = `insight 语义:lesson(曾踩坑/纠偏, pattern→fix) | decision(权衡选型, choice→reason) | procedure(多步指南, steps) | experience(problem→solution,v0.4 兼容)。默认 lesson。所有 kind 都可带 trigger:命中即在动手前确定性注入。`
|
|
15
15
|
const scopeDesc =
|
|
16
16
|
'作用域:task(任务私有,随任务归档,不进共享注入) | project(项目资产) | global(个人能力库,跨项目,建议配 trigger 关键词以便将来技能命中)。' +
|
|
17
17
|
'解析顺序:显式 scope > task_id > 当前绑定任务 > project。'
|
|
@@ -37,9 +37,14 @@ export function lessonTool(config) {
|
|
|
37
37
|
properties: {
|
|
38
38
|
keywords: { type: 'array', items: { type: 'string' } },
|
|
39
39
|
symbols: { type: 'array', items: { type: 'string' } },
|
|
40
|
+
actions: { type: 'array', items: { type: 'string' } },
|
|
41
|
+
paths: { type: 'array', items: { type: 'string' } },
|
|
40
42
|
scope: { type: 'array', items: { type: 'string' } },
|
|
41
43
|
},
|
|
42
|
-
description:
|
|
44
|
+
description:
|
|
45
|
+
'authored trigger(所有 kind 通用):命中即在**动手前**确定性注入本条。' +
|
|
46
|
+
'keywords/symbols 匹配人类消息与工具参数;actions 用归一 id(git-commit / npm-publish / go-public / deploy / delete / migrate …);' +
|
|
47
|
+
'paths 用 glob(README* / CHANGELOG* / .gitignore);scope 按项目画像 tags 过滤。留空则只可能作为统计提示注入。',
|
|
43
48
|
},
|
|
44
49
|
task_id: { type: 'string', description: '目标任务 id(scope 缺省时优先于绑定任务)' },
|
|
45
50
|
files: { type: 'array', items: { type: 'string' }, description: '关联文件(项目相对路径)' },
|