@yolk_vat-y/dsh-project-memory 0.5.6 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +46 -0
- package/README.md +1 -1
- package/README.zh-CN.md +1 -1
- package/package.json +2 -2
- package/src/auto-inject.js +346 -281
- package/src/chunker.js +14 -6
- package/src/commands/insight-actions.js +18 -14
- package/src/commands/invocation.js +36 -0
- package/src/commands/task-actions.js +24 -39
- package/src/commands/tasks.js +18 -13
- package/src/enhancer.js +11 -2
- package/src/index-pipeline.js +119 -0
- package/src/insight-store.js +57 -28
- package/src/lazy.js +15 -46
- package/src/link.js +10 -1
- package/src/parsers/pdfjs-parser.js +2 -4
- package/src/project-profile.js +12 -5
- package/src/readiness.js +27 -3
- package/src/recall.js +28 -8
- package/src/setup/taskbridge.js +11 -9
- package/src/store.js +56 -17
- package/src/symbols.js +58 -19
- package/src/tools/forget.js +2 -1
- package/src/tools/index-doc.js +4 -1
- package/src/tools/index-repo.js +33 -86
- package/src/tools/lesson-tools.js +2 -1
- package/src/tools/query-memory.js +17 -3
- package/src/tools/remember.js +3 -1
- package/src/tools/task-tools.js +4 -1
- package/src/tools/watch-repo.js +9 -5
- package/src/util/fs.js +14 -0
- package/src/util/session-cache.js +72 -0
- package/src/watch.js +39 -81
package/src/lazy.js
CHANGED
|
@@ -1,11 +1,8 @@
|
|
|
1
1
|
import path from 'node:path'
|
|
2
2
|
import { tmpdir } from 'node:os'
|
|
3
3
|
import { existsSync, readdirSync, statSync } from 'node:fs'
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
6
|
-
import { docEntriesNeedBackfill } from './doc-index.js'
|
|
7
|
-
import { scanSymbols } from './symbols.js'
|
|
8
|
-
import { linkEntries } from './link.js'
|
|
4
|
+
import { memoryRootFor, relativePath, storeKey } from './util/fs.js'
|
|
5
|
+
import { DROP, OVERSIZE, UNCHANGED, commitFileUpdates, fileKind, planFileIndex, toFileUpdate } from './index-pipeline.js'
|
|
9
6
|
import { ProjectMemoryStore } from './store.js'
|
|
10
7
|
import { onFileObserved } from './enhancer.js'
|
|
11
8
|
|
|
@@ -78,70 +75,42 @@ export function findProjectRoot(filePath, ceiling = path.resolve(tmpdir())) {
|
|
|
78
75
|
}
|
|
79
76
|
|
|
80
77
|
export async function indexFile(ctx, config, filePath, watchManager = null) {
|
|
81
|
-
const
|
|
82
|
-
if (!
|
|
78
|
+
const kind = fileKind(path.extname(filePath).toLowerCase())
|
|
79
|
+
if (!kind) return false
|
|
83
80
|
const root = findProjectRoot(filePath)
|
|
84
81
|
if (!root) return false
|
|
85
82
|
|
|
86
83
|
const memoryDir = memoryRootFor(root, config.memoryDir)
|
|
87
84
|
const store = new ProjectMemoryStore(memoryDir).load()
|
|
88
85
|
const rel = storeKey(relativePath(root, filePath))
|
|
89
|
-
const
|
|
90
|
-
|
|
86
|
+
const record = store.fileRecord(rel)
|
|
87
|
+
|
|
91
88
|
let size
|
|
92
|
-
let
|
|
93
|
-
let entries
|
|
89
|
+
let plan
|
|
94
90
|
try {
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
}
|
|
98
|
-
;({ hash, size, buffer } = readFileForIndex(filePath))
|
|
91
|
+
size = statSync(filePath).size
|
|
92
|
+
plan = await planFileIndex({ rel, filePath, kind, config, record, existingEntries: store.entries[rel], size })
|
|
99
93
|
} catch {
|
|
100
94
|
return false
|
|
101
95
|
}
|
|
102
|
-
|
|
103
|
-
if (existing && existing.sha256 === hash && !(isSupportedDoc(ext) && docEntriesNeedBackfill(store.entries[rel]))) return false
|
|
96
|
+
if (plan.key === UNCHANGED || plan.key === OVERSIZE) return false
|
|
104
97
|
|
|
105
98
|
if (watchManager) {
|
|
106
99
|
watchManager.addRoot(root)
|
|
107
100
|
store.addWatch(root)
|
|
108
101
|
}
|
|
109
102
|
|
|
110
|
-
if (
|
|
111
|
-
entries = scanSymbols(rel, filePath, buffer.toString('utf8'))
|
|
103
|
+
if (plan.key !== DROP && plan.type === 'code') {
|
|
112
104
|
onFileObserved(store, rel, filePath, config, root)
|
|
113
|
-
return store.commit((s) => {
|
|
114
|
-
s.markFile(rel, { sha256: hash, size, type: 'code', indexedAt: new Date().toISOString() })
|
|
115
|
-
s.setEntries(rel, entries)
|
|
116
|
-
linkEntries(s)
|
|
117
|
-
return true
|
|
118
|
-
})
|
|
119
|
-
} else {
|
|
120
|
-
entries = await buildDocEntries(rel, filePath, {
|
|
121
|
-
chunkChars: config.chunkChars,
|
|
122
|
-
maxChunks: config.maxChunksPerFile,
|
|
123
|
-
maxFileSizeMb: config.maxFileSizeMb,
|
|
124
|
-
maxPdfPages: config.maxPdfPages,
|
|
125
|
-
})
|
|
126
|
-
if (entries === null) {
|
|
127
|
-
return store.commit((s) => {
|
|
128
|
-
s.removeFile(rel)
|
|
129
|
-
return false
|
|
130
|
-
})
|
|
131
|
-
}
|
|
132
|
-
return store.commit((s) => {
|
|
133
|
-
s.markFile(rel, { sha256: hash, size, type: 'doc', indexedAt: new Date().toISOString() })
|
|
134
|
-
s.setEntries(rel, entries)
|
|
135
|
-
linkEntries(s)
|
|
136
|
-
return true
|
|
137
|
-
})
|
|
138
105
|
}
|
|
106
|
+
commitFileUpdates(store, { updates: [toFileUpdate(rel, plan, record)] })
|
|
107
|
+
return plan.key !== DROP
|
|
139
108
|
}
|
|
140
109
|
|
|
141
110
|
export function codeFirst(paths) {
|
|
142
111
|
return [...paths].sort((a, b) => {
|
|
143
|
-
const aCode =
|
|
144
|
-
const bCode =
|
|
112
|
+
const aCode = fileKind(path.extname(a).toLowerCase()) === 'code' ? 0 : 1
|
|
113
|
+
const bCode = fileKind(path.extname(b).toLowerCase()) === 'code' ? 0 : 1
|
|
145
114
|
return aCode - bCode
|
|
146
115
|
})
|
|
147
116
|
}
|
package/src/link.js
CHANGED
|
@@ -39,7 +39,16 @@ export function linkEntries(store) {
|
|
|
39
39
|
if (linked.size > before) links++
|
|
40
40
|
}
|
|
41
41
|
}
|
|
42
|
-
|
|
42
|
+
const before = Array.isArray(doc.linkedSymbols) ? doc.linkedSymbols.join('\u0000') : ''
|
|
43
|
+
const next = [...linked]
|
|
44
|
+
if (before === next.join('\u0000')) continue
|
|
45
|
+
doc.linkedSymbols = next.length ? next : undefined
|
|
46
|
+
// 链接是在**符号**落盘那一刻算出来的,此时 doc 的 shard 往往不是脏的;不标脏就只存在于内存,
|
|
47
|
+
// 下次进程启动重新加载后链接全部丢失("文档先索引、符号后到"的正常顺序)。
|
|
48
|
+
if (typeof store.markFile === 'function' && typeof store.fileRecord === 'function' && doc.sourcePath) {
|
|
49
|
+
const record = store.fileRecord(doc.sourcePath)
|
|
50
|
+
if (record) store.markFile(doc.sourcePath, record)
|
|
51
|
+
}
|
|
43
52
|
}
|
|
44
53
|
return links
|
|
45
54
|
}
|
|
@@ -60,9 +60,8 @@ export async function parsePdf(filePath, { pages = null, maxPages = 1000, backen
|
|
|
60
60
|
const data = new Uint8Array(await readFile(filePath))
|
|
61
61
|
const { getDocument } = await loadPdfjs()
|
|
62
62
|
const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
|
|
63
|
-
const doc = await loadingTask.promise
|
|
64
|
-
|
|
65
63
|
try {
|
|
64
|
+
const doc = await loadingTask.promise
|
|
66
65
|
const total = doc.numPages
|
|
67
66
|
if (total > maxPages) {
|
|
68
67
|
throw new Error(`PDF has ${total} pages, over the maxPages limit of ${maxPages}`)
|
|
@@ -112,9 +111,8 @@ export async function parsePdfInfo(filePath, maxPages = 1000) {
|
|
|
112
111
|
const data = new Uint8Array(await readFile(filePath))
|
|
113
112
|
const { getDocument } = await loadPdfjs()
|
|
114
113
|
const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
|
|
115
|
-
const doc = await loadingTask.promise
|
|
116
|
-
|
|
117
114
|
try {
|
|
115
|
+
const doc = await loadingTask.promise
|
|
118
116
|
if (doc.numPages > maxPages) {
|
|
119
117
|
throw new Error(`PDF has ${doc.numPages} pages, over the maxPages limit of ${maxPages}`)
|
|
120
118
|
}
|
package/src/project-profile.js
CHANGED
|
@@ -17,11 +17,18 @@ function depsOf(manifest, fields) {
|
|
|
17
17
|
const tags = new Set()
|
|
18
18
|
for (const f of fields) {
|
|
19
19
|
const deps = manifest && manifest[f]
|
|
20
|
-
if (deps
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
20
|
+
if (!deps || typeof deps !== 'object') continue
|
|
21
|
+
for (const name of Object.keys(deps)) {
|
|
22
|
+
if (typeof name !== 'string' || !name) continue
|
|
23
|
+
tags.add(name.toLowerCase())
|
|
24
|
+
if (!name.startsWith('@')) continue
|
|
25
|
+
// scoped 包名 `@scope/pkg`:要加的是 **scope**(`vue`),不是包名段(`compiler-sfc`)。
|
|
26
|
+
// 旧写法 `name.split('/')[1]` 让 `@vue/*` 项目永远拿不到 `vue` tag,
|
|
27
|
+
// trigger.scope:['vue'] / legacy scope 过滤会把最相关的那条 procedure 静默判死;
|
|
28
|
+
// 而 `@malformed`(没有 `/`)会在这里抛错,异常被 collectTags 整个吞掉 → 全项目 tags 清零。
|
|
29
|
+
const parts = name.slice(1).split('/')
|
|
30
|
+
if (parts[0]) tags.add(parts[0].toLowerCase())
|
|
31
|
+
if (parts[1]) tags.add(parts[1].toLowerCase())
|
|
25
32
|
}
|
|
26
33
|
}
|
|
27
34
|
return tags
|
package/src/readiness.js
CHANGED
|
@@ -82,7 +82,16 @@ export function buildReadinessContext(input = {}) {
|
|
|
82
82
|
const humanText = String(input.humanText ?? input.query ?? '')
|
|
83
83
|
const actionText = String(input.actionText || '')
|
|
84
84
|
const actions = new Set([...(input.actions || []), ...detectActions(humanText), ...detectActions(actionText)])
|
|
85
|
-
|
|
85
|
+
// 人类消息里提到的路径 = **动手前的写意图**("帮我改 src/x.js");动作文本里的路径不区分读写,
|
|
86
|
+
// 读一个文件不该算写目标(否则 `read README.md` 会命中 writes:['README.md'])。
|
|
87
|
+
// 两者对外仍合并成 `paths`(兼容既有语义),写通道单看 humanPaths。
|
|
88
|
+
// 调用方若已带 humanPaths(injectForStep 构建后再传给 buildInjection),以它为准,
|
|
89
|
+
// 否则退回 paths —— 避免把上一次算出来的并集(含读路径)再当写意图。
|
|
90
|
+
const humanPaths = new Set([
|
|
91
|
+
...(Array.isArray(input.humanPaths) ? input.humanPaths : (input.paths || [])),
|
|
92
|
+
...extractPaths(humanText),
|
|
93
|
+
])
|
|
94
|
+
const paths = new Set([...humanPaths, ...extractPaths(actionText)])
|
|
86
95
|
// 动作平面(S1):结构化调用优先;没有结构化调用时对 actionText 跑一遍 shell 规则兜底。
|
|
87
96
|
const fromText = activityFromText(actionText)
|
|
88
97
|
const ops = new Set([...(input.ops || []), ...fromText.ops])
|
|
@@ -97,6 +106,7 @@ export function buildReadinessContext(input = {}) {
|
|
|
97
106
|
actionText,
|
|
98
107
|
actions: [...actions],
|
|
99
108
|
paths: [...paths],
|
|
109
|
+
humanPaths: [...humanPaths],
|
|
100
110
|
ops: [...ops],
|
|
101
111
|
targets: [...targets],
|
|
102
112
|
hosts: [...hosts],
|
|
@@ -209,6 +219,10 @@ export function intentText(text) {
|
|
|
209
219
|
.replace(/“[^”\n]*”/g, ' ')
|
|
210
220
|
.replace(/[A-Za-z]:\\[^\s"']*/g, ' ')
|
|
211
221
|
.replace(/(?:[A-Za-z0-9_.@-]+\/)+[A-Za-z0-9_.@-]+/g, ' ')
|
|
222
|
+
// 带扩展名的**非 ASCII 文件名**(`石啸天-记忆方向调研.pptx` / `docs/面试演示-王金鹏.md`):
|
|
223
|
+
// 只挡 ASCII 路径不够——中文文件名整块留下,"调研""面试"这类子串照样触发 when.intents,
|
|
224
|
+
// 正是引号剥离要防的那类假阳性。ASCII 裸名(如 de-TODO.md)保持原语义,不在这里动。
|
|
225
|
+
.replace(/\S*[^\x00-\x7F]\S*\.[A-Za-z][A-Za-z0-9]{0,7}\b/g, ' ')
|
|
212
226
|
// 仓库里的裸文件名(README / CHANGELOG …)也是语料,不是意图
|
|
213
227
|
.replace(/\b(?:README|CHANGELOG|LICENSE|AGENTS|CONTRIBUTING|Dockerfile|Makefile)\b/g, ' ')
|
|
214
228
|
.replace(/\s+/g, ' ')
|
|
@@ -301,10 +315,14 @@ export function matchTrigger(trigger, ctx) {
|
|
|
301
315
|
for (const op of when.ops || []) {
|
|
302
316
|
if (op && (ctx?.ops || []).includes(String(op))) return `op:${op}`
|
|
303
317
|
}
|
|
318
|
+
// 写目标 = 已观察到的结构化写调用 ∪ 人类消息里提到的路径(动手前的写意图)。
|
|
319
|
+
// DSH 没有工具执行前拦截钩子,动手前唯一能拿到的写意图就是人类消息里的路径;
|
|
320
|
+
// 但动作文本里的**读**路径不算(`read README.md` 不是"即将写 README.md")。
|
|
321
|
+
const writeTargets = [...(ctx?.targets || []), ...(ctx?.humanPaths || [])]
|
|
304
322
|
for (const w of when.writes || []) {
|
|
305
323
|
// 扩展名/泛名 glob 在这里被硬性忽略:`*.pptx` 这类条件只能撒谎,不能收窄。
|
|
306
324
|
if (!w || isDroppableGlob(w)) continue
|
|
307
|
-
if (matchAnyPath(w,
|
|
325
|
+
if (matchAnyPath(w, writeTargets)) return `write:${w}`
|
|
308
326
|
}
|
|
309
327
|
const intent = ctx?.intent ?? ctx?.humanText ?? ''
|
|
310
328
|
for (const it of when.intents || []) {
|
|
@@ -355,6 +373,9 @@ export function normalizeTrigger(it, opts = {}) {
|
|
|
355
373
|
// (实测:那样会让 D 场景一次多出 6 条假阳性)。
|
|
356
374
|
const intents = []
|
|
357
375
|
for (const k of t.keywords || []) if (isIntentWord(k)) intents.push(String(k))
|
|
376
|
+
// 旧 `symbols` 也走文本平面:不归一就等于静默丢掉一个作者写下的触发面
|
|
377
|
+
// (README 承诺 keywords/symbols/actions/paths/scope 都会迁移)。与 keywords 同一把准入尺子。
|
|
378
|
+
for (const s of t.symbols || []) if (isIntentWord(s)) intents.push(String(s))
|
|
358
379
|
const when = {}
|
|
359
380
|
if (ops.size) when.ops = [...ops].sort()
|
|
360
381
|
if (writes.length) when.writes = [...new Set(writes)]
|
|
@@ -463,6 +484,9 @@ export function relativeHits(scored, { ratioMin = 0.5 } = {}) {
|
|
|
463
484
|
const list = (scored || []).filter((r) => r && Number.isFinite(r.score) && r.score > 0)
|
|
464
485
|
if (!list.length) return []
|
|
465
486
|
const top = list[0].score
|
|
466
|
-
|
|
487
|
+
// 显式的 0 表示"关掉相对门槛"(只剩 score>0);未给/非法值才回退 0.5。
|
|
488
|
+
// 旧写法 `ratioMin > 0 ? ratioMin : 0.5` 把 0 当成"没配",配置上无法关闭。
|
|
489
|
+
const ratio = typeof ratioMin === 'number' && Number.isFinite(ratioMin) ? ratioMin : 0.5
|
|
490
|
+
const floor = top * Math.max(0, Math.min(1, ratio))
|
|
467
491
|
return list.filter((r) => r.score >= floor)
|
|
468
492
|
}
|
package/src/recall.js
CHANGED
|
@@ -101,6 +101,21 @@ export function insightToEntry(ins) {
|
|
|
101
101
|
}
|
|
102
102
|
}
|
|
103
103
|
|
|
104
|
+
/**
|
|
105
|
+
* 一条 insight 的**打分文本**(BM25 与 IDF 覆盖率必须用同一份)。
|
|
106
|
+
*
|
|
107
|
+
* 为什么要有这个函数:提示通道的"这条查询离语料有多远"(`idfCoverage` 的 df)曾经用
|
|
108
|
+
* `insightMatchText` 计算,而排序用 `insightToEntry`(含 fix/solution/reason/steps)。
|
|
109
|
+
* 于是"只在解药里有匹配"的查询词 df=0 → supportRatio=0 → **整条提示通道本轮沉默**,
|
|
110
|
+
* 而排序明明给了它高分。两份文本永远来自同一个 entry,才不会再漂移。
|
|
111
|
+
* @param {object} ins
|
|
112
|
+
* @returns {string}
|
|
113
|
+
*/
|
|
114
|
+
export function insightScoringText(ins) {
|
|
115
|
+
const e = insightToEntry(ins)
|
|
116
|
+
return [e.title, e.summary, e.terms, ...(e.keywords || [])].filter(Boolean).join('\n')
|
|
117
|
+
}
|
|
118
|
+
|
|
104
119
|
/**
|
|
105
120
|
* 召回可见的 insight 集合:作用域可见性跟随会话绑定。
|
|
106
121
|
* - project / global:始终可见(归档除外);
|
|
@@ -127,6 +142,9 @@ export function visibleInsights({ store = null, globalStore = null, boundTaskId
|
|
|
127
142
|
const task = store.getTask(boundTaskId)
|
|
128
143
|
for (const it of (task && task.insights) || []) {
|
|
129
144
|
if (!it || it.archived) continue
|
|
145
|
+
// 任务级也要挡草稿:reflection 写的是 draft:true,未确认前不该被 query_memory 召回
|
|
146
|
+
// (project/global 一直有这道闸,任务级漏了)。
|
|
147
|
+
if (it.draft === true) continue
|
|
130
148
|
out.push({ ...it, scope: it.scope || 'task' })
|
|
131
149
|
}
|
|
132
150
|
}
|
|
@@ -163,15 +181,17 @@ export function recallItems(opts = {}) {
|
|
|
163
181
|
if (want.has('doc') || want.has('symbol')) {
|
|
164
182
|
const pool = (entries || (store && typeof store.allEntries === 'function' ? store.allEntries() : []) || [])
|
|
165
183
|
.filter((e) => e && (e.type === 'doc' || e.type === 'symbol') && want.has(e.type))
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
for (const { entry, score } of scored) {
|
|
169
|
-
const prior = entry.type === 'doc' ? normativePrior(entry) : LAYER_PRIORS.symbol
|
|
170
|
-
byLayer.get(entry.type).push({ item: entry, score, prior, weightedScore: score * prior })
|
|
171
|
-
}
|
|
184
|
+
// 每层各自 top-k(不变量 #2):一次跨类型取 top-k 会让文档把符号挤出结果——
|
|
185
|
+
// 20 条命中文档 + 1 条命中符号、limit 8 时符号层直接消失。分类型各取一次再分桶。
|
|
172
186
|
for (const layer of ['doc', 'symbol']) {
|
|
173
|
-
const
|
|
174
|
-
if (!
|
|
187
|
+
const layerPool = pool.filter((e) => e.type === layer)
|
|
188
|
+
if (!layerPool.length) continue
|
|
189
|
+
const scored = rankEntriesStreaming(layerPool, qs, idf, limit)
|
|
190
|
+
if (!scored.length) continue
|
|
191
|
+
const hits = scored.map(({ entry, score }) => {
|
|
192
|
+
const prior = layer === 'doc' ? normativePrior(entry) : LAYER_PRIORS.symbol
|
|
193
|
+
return { item: entry, score, prior, weightedScore: score * prior }
|
|
194
|
+
})
|
|
175
195
|
hits.sort((a, b) => b.weightedScore - a.weightedScore)
|
|
176
196
|
const top = hits[0].weightedScore || 1
|
|
177
197
|
buckets.push({
|
package/src/setup/taskbridge.js
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
// 事实依据:dsh session/event 签名 (session, event);todo/write data.todos;
|
|
3
3
|
// tool/call data.arguments 为 JSON 字符串;fs 工具名 read/write/edit/read_image,参数 file_path。
|
|
4
4
|
import path from 'node:path'
|
|
5
|
-
import { createHash } from 'node:crypto'
|
|
5
|
+
import { createHash, randomUUID } from 'node:crypto'
|
|
6
6
|
import { memoryRootFor } from '../util/fs.js'
|
|
7
7
|
import { findProjectRoot } from '../lazy.js'
|
|
8
8
|
import { ProjectMemoryStore } from '../store.js'
|
|
@@ -27,7 +27,9 @@ export function slugifyTitle(text) {
|
|
|
27
27
|
|
|
28
28
|
export function genTaskId(projectRoot, title) {
|
|
29
29
|
const slug = slugifyTitle(title) || 'task'
|
|
30
|
-
|
|
30
|
+
// 毫秒不是唯一性保证:两个会话在同一毫秒建同名任务会拿到同一个 id,
|
|
31
|
+
// addTask 变成"同一 id 两次",两个会话的绑定互相串台。补 6 位随机后缀。
|
|
32
|
+
return `tsk_${hash8(projectRoot)}_${slug}_${Date.now().toString(36)}_${randomUUID().slice(0, 6)}`
|
|
31
33
|
}
|
|
32
34
|
|
|
33
35
|
/** 项目根推导:findProjectRoot 期望文件路径,传目录会从父级起跳,故用目录内探针路径。 */
|
|
@@ -138,13 +140,7 @@ export function hotSortFiles(files, meta) {
|
|
|
138
140
|
/** 触碰文件后更新元数据并重排 task.files(写 vs 读分别记时间/次数)。 */
|
|
139
141
|
export function touchTaskFile(task, rel, kind, now) {
|
|
140
142
|
task.fileMeta = task.fileMeta || {}
|
|
141
|
-
if (!task.files.includes(rel))
|
|
142
|
-
task.files.push(rel)
|
|
143
|
-
if (task.files.length > MAX_FILES_PER_TASK) {
|
|
144
|
-
const dropped = task.files.shift()
|
|
145
|
-
delete task.fileMeta[dropped]
|
|
146
|
-
}
|
|
147
|
-
}
|
|
143
|
+
if (!task.files.includes(rel)) task.files.push(rel)
|
|
148
144
|
const m = task.fileMeta[rel] || {}
|
|
149
145
|
m.n = (m.n || 0) + 1 // 兼容旧字段:读+写总数(保留,避免老任务语义漂移)
|
|
150
146
|
// 读/写分计数(先只采集,不消费 —— 排序与注入仍只用 lastWriteAt/lastReadAt)
|
|
@@ -157,6 +153,12 @@ export function touchTaskFile(task, rel, kind, now) {
|
|
|
157
153
|
m.lastReadAt = now
|
|
158
154
|
task.fileMeta[rel] = m
|
|
159
155
|
task.files = hotSortFiles(task.files, task.fileMeta)
|
|
156
|
+
// 超限丢最冷的:列表已按热度降序,末尾即最冷。旧代码在**重排前** shift(),
|
|
157
|
+
// 而那时 files 本身就是热序,shift 恰好把最热的那个文件(刚写的)丢掉、留下最冷的。
|
|
158
|
+
while (task.files.length > MAX_FILES_PER_TASK) {
|
|
159
|
+
const dropped = task.files.pop()
|
|
160
|
+
delete task.fileMeta[dropped]
|
|
161
|
+
}
|
|
160
162
|
}
|
|
161
163
|
|
|
162
164
|
/**
|
package/src/store.js
CHANGED
|
@@ -17,6 +17,14 @@ const SHARDS_DIR = 'shards'
|
|
|
17
17
|
const storeCache = new Map()
|
|
18
18
|
const STORE_CACHE_MAX = 32
|
|
19
19
|
|
|
20
|
+
/** 已就"无法迁移的旧 store"告警过的目录:避免每次 load() 都刷一行。 */
|
|
21
|
+
const migrationWarned = new Set()
|
|
22
|
+
|
|
23
|
+
/** 纯对象判定(排除 null / 数组):磁盘读入的 JSON 形状校验统一走它。 */
|
|
24
|
+
function isRecord(value) {
|
|
25
|
+
return Boolean(value) && typeof value === 'object' && !Array.isArray(value)
|
|
26
|
+
}
|
|
27
|
+
|
|
20
28
|
function loadJson(filePath, fallback) {
|
|
21
29
|
let raw
|
|
22
30
|
try {
|
|
@@ -82,7 +90,9 @@ export class ProjectMemoryStore {
|
|
|
82
90
|
load() {
|
|
83
91
|
const key = path.resolve(this.dir)
|
|
84
92
|
const hot = storeCache.get(key)
|
|
85
|
-
|
|
93
|
+
// 缓存命中即返回:`hot === this` 时再读一遍盘会静默丢弃本实例尚未 save() 的变更
|
|
94
|
+
// (_loadSharded/_loadInsights 会重新赋值 experience/tasks/insights…)。
|
|
95
|
+
if (hot) return hot
|
|
86
96
|
this._migrateLegacyIfNeeded()
|
|
87
97
|
this._loadSharded()
|
|
88
98
|
this._loadInsights()
|
|
@@ -114,9 +124,34 @@ export class ProjectMemoryStore {
|
|
|
114
124
|
}
|
|
115
125
|
const legacyEntriesPath = path.join(this.dir, ENTRIES_FILE)
|
|
116
126
|
if (!existsSafe(legacyEntriesPath)) return
|
|
117
|
-
const
|
|
118
|
-
|
|
119
|
-
const
|
|
127
|
+
const legacyIndexPath = path.join(this.dir, INDEX_FILE)
|
|
128
|
+
// 先读两份旧文件:损坏的那份会在这里被 loadJson 备份为 .corrupt(保留现场)。
|
|
129
|
+
const index = loadJson(legacyIndexPath, null)
|
|
130
|
+
const entries = loadJson(legacyEntriesPath, null)
|
|
131
|
+
if (!isRecord(index) || !isRecord(entries)) {
|
|
132
|
+
const hasEntries = isRecord(entries) && Object.keys(entries).length > 0
|
|
133
|
+
if (!hasEntries) {
|
|
134
|
+
// entries 损坏(已备份)或本就是空的:没有可保护的数据,按空旧库收尾。
|
|
135
|
+
writeJsonAtomic(formatPath, { version: 2, layout: 'sharded' })
|
|
136
|
+
for (const stale of [legacyEntriesPath, legacyIndexPath]) {
|
|
137
|
+
try {
|
|
138
|
+
unlinkSync(stale)
|
|
139
|
+
} catch {
|
|
140
|
+
// already renamed away by corrupt backup, or gone; nothing to do
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
return
|
|
144
|
+
}
|
|
145
|
+
// index.json 是 entries.json → rel 的唯一映射。它缺失或损坏时继续迁移,会写出 0 个
|
|
146
|
+
// shard、打上 v2 标记、再把**完好的** entries.json 删掉——等于一次静默的数据清空。
|
|
147
|
+
// 保留现场,不写 format 标记,等 index.json 修好后再迁(每次进程只提示一次)。
|
|
148
|
+
if (!migrationWarned.has(this.dir)) {
|
|
149
|
+
migrationWarned.add(this.dir)
|
|
150
|
+
console.error(`[dsh-project-memory] legacy store at ${this.dir} has ${ENTRIES_FILE} but no readable ${INDEX_FILE}; migration skipped to protect it`)
|
|
151
|
+
}
|
|
152
|
+
return
|
|
153
|
+
}
|
|
154
|
+
const files = isRecord(index.files) ? index.files : {}
|
|
120
155
|
const orphans = Object.keys(entries).filter((rel) => !(rel in files))
|
|
121
156
|
if (orphans.length) {
|
|
122
157
|
console.error(
|
|
@@ -147,9 +182,11 @@ export class ProjectMemoryStore {
|
|
|
147
182
|
}
|
|
148
183
|
for (const name of shardNames) {
|
|
149
184
|
const shard = loadJson(path.join(this.dir, SHARDS_DIR, name), null)
|
|
150
|
-
if (!shard || typeof shard.relPath !== 'string' || !shard.record) continue
|
|
185
|
+
if (!shard || typeof shard.relPath !== 'string' || !isRecord(shard.record)) continue
|
|
151
186
|
this.files[shard.relPath] = shard.record
|
|
152
|
-
|
|
187
|
+
// 畸形 shard(entries 被写成对象/null)不能让 allEntries() 在 `for…of` 上抛错,
|
|
188
|
+
// 否则一个坏文件会拖垮整个进程的每一次读取。
|
|
189
|
+
this.entries[shard.relPath] = Array.isArray(shard.entries) ? shard.entries.filter(isRecord) : []
|
|
153
190
|
}
|
|
154
191
|
this.experience = loadJson(path.join(this.dir, EXPERIENCE_FILE), [])
|
|
155
192
|
this.tasks = loadJson(path.join(this.dir, TASKS_FILE), [])
|
|
@@ -165,7 +202,10 @@ export class ProjectMemoryStore {
|
|
|
165
202
|
// migratedAt 落盘保证跨进程/崩溃幂等。销毁式收敛放到 recall 统一 PR。
|
|
166
203
|
_loadInsights() {
|
|
167
204
|
const doc = loadJson(path.join(this.dir, INSIGHTS_FILE), null)
|
|
168
|
-
this.insights = doc &&
|
|
205
|
+
this.insights = isRecord(doc) && Array.isArray(doc.items)
|
|
206
|
+
// items 里混进 null/非对象(手改或旧版写入)会让迁移与召回逐个 `.title` 抛错——过滤掉。
|
|
207
|
+
? { ...doc, items: doc.items.filter(isRecord) }
|
|
208
|
+
: { version: 1, migratedAt: null, items: [] }
|
|
169
209
|
this._migrateExperienceToInsights()
|
|
170
210
|
// PR3:v1 → v2 懒回填派生 trigger(纯确定性、幂等;不调用模型,不改写已有字段)
|
|
171
211
|
if (backfillDerivedTriggers(this.insights)) this._dirtyInsights = true
|
|
@@ -551,15 +591,14 @@ export class ProjectMemoryStore {
|
|
|
551
591
|
return result
|
|
552
592
|
}
|
|
553
593
|
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
}
|
|
594
|
+
/**
|
|
595
|
+
* 写入一条文件更新,以 expectedHash 做 CAS。
|
|
596
|
+
* @returns {boolean} true = 已写入;false = 文件自扫描后又被改动,本次拒绝
|
|
597
|
+
* (调用方让该条目保持未落快照,下一轮重试)。
|
|
598
|
+
* 移除条目不经过这里:调用方在 commit 里直接 removeFile(见 index-pipeline.js)。
|
|
599
|
+
*/
|
|
600
|
+
applyFileUpdate(relPath, { expectedHash, hash, entries, type, size }) {
|
|
601
|
+
if ((this.fileRecord(relPath)?.sha256 ?? null) !== (expectedHash ?? null)) return false
|
|
563
602
|
this.markFile(relPath, {
|
|
564
603
|
sha256: hash,
|
|
565
604
|
size,
|
|
@@ -567,7 +606,7 @@ export class ProjectMemoryStore {
|
|
|
567
606
|
indexedAt: new Date().toISOString(),
|
|
568
607
|
})
|
|
569
608
|
this.setEntries(relPath, entries)
|
|
570
|
-
return
|
|
609
|
+
return true
|
|
571
610
|
}
|
|
572
611
|
}
|
|
573
612
|
|
package/src/symbols.js
CHANGED
|
@@ -144,18 +144,57 @@ function extractTypeSignature(line) {
|
|
|
144
144
|
return sig
|
|
145
145
|
}
|
|
146
146
|
|
|
147
|
-
function
|
|
148
|
-
|
|
149
|
-
const
|
|
150
|
-
|
|
151
|
-
|
|
147
|
+
function braceDelta(s) {
|
|
148
|
+
let d = 0
|
|
149
|
+
for (const ch of s) {
|
|
150
|
+
if (ch === '{') d++
|
|
151
|
+
else if (ch === '}') d--
|
|
152
152
|
}
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
153
|
+
return d
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* 读取一条 interface / type 声明,支持 `export`(含 `declare`)与多行形态。
|
|
158
|
+
* 旧实现要求 `^interface`(不吃 export)且 `}` 必须在本行,于是
|
|
159
|
+
* `export interface X {…}`、任何多行 interface、`export type X = …` 在 L1 正则扫描器里
|
|
160
|
+
* 全都产出 0 个符号(TS 增强器可用时才被补回来;没有 typescript 的项目就彻底看不到)。
|
|
161
|
+
* @returns {{kind: 'interface'|'type', name: string, text: string, endIdx: number} | null}
|
|
162
|
+
*/
|
|
163
|
+
function readTypeDeclaration(rawLines, startIdx) {
|
|
164
|
+
const first = rawLines[startIdx].trim()
|
|
165
|
+
const m = first.match(/^(?:export\s+)?(?:declare\s+)?(interface|type)\s+([A-Za-z_$][\w$]*)/)
|
|
166
|
+
if (!m) return null
|
|
167
|
+
const kind = m[1]
|
|
168
|
+
const parts = [first]
|
|
169
|
+
let depth = braceDelta(first)
|
|
170
|
+
let endIdx = startIdx
|
|
171
|
+
const complete = () => (kind === 'interface'
|
|
172
|
+
? depth <= 0 && parts[parts.length - 1].includes('}')
|
|
173
|
+
: depth <= 0 && /[;}]/.test(parts[parts.length - 1]))
|
|
174
|
+
for (let j = startIdx + 1; j < rawLines.length && j - startIdx <= 20 && !complete(); j++) {
|
|
175
|
+
const t = rawLines[j].trim()
|
|
176
|
+
if (!t) {
|
|
177
|
+
if (depth <= 0) break
|
|
178
|
+
parts.push(t)
|
|
179
|
+
endIdx = j
|
|
180
|
+
continue
|
|
181
|
+
}
|
|
182
|
+
// 归零后遇到下一条声明开头就收尾(TS 的 type 别名常不写分号)
|
|
183
|
+
if (depth <= 0 && j > startIdx && /^(?:export\s+)?(?:declare\s+)?(?:interface|type|function|class|const|let|var|import|enum)\b/.test(t)) break
|
|
184
|
+
parts.push(t)
|
|
185
|
+
depth += braceDelta(t)
|
|
186
|
+
endIdx = j
|
|
157
187
|
}
|
|
158
|
-
return ''
|
|
188
|
+
return { kind, name: m[2], text: parts.join(' '), endIdx }
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
function formatTypeDeclaration({ kind, text }) {
|
|
192
|
+
if (kind === 'interface') {
|
|
193
|
+
const body = text.match(/\{([\s\S]*)\}/)
|
|
194
|
+
return body ? `{ ${body[1].replace(/\s+/g, ' ').trim()} }` : ''
|
|
195
|
+
}
|
|
196
|
+
const eq = text.match(/=\s*([\s\S]+?);?\s*$/)
|
|
197
|
+
return eq ? `= ${eq[1].replace(/\s+/g, ' ').trim()}` : ''
|
|
159
198
|
}
|
|
160
199
|
|
|
161
200
|
function extractOverloads(masked, startIdx) {
|
|
@@ -226,6 +265,7 @@ function scanJsLike(masked, relPath, rawLines) {
|
|
|
226
265
|
const symbols = []
|
|
227
266
|
let prevOpensBlock = false
|
|
228
267
|
for (let i = 0; i < masked.length; i++) {
|
|
268
|
+
const declStart = i // 多行声明会推进 i,符号行号要记声明的**首行**
|
|
229
269
|
const rawText = rawLines[i].trim()
|
|
230
270
|
const maskedText = masked[i].trim()
|
|
231
271
|
if (!rawText) continue
|
|
@@ -233,13 +273,12 @@ function scanJsLike(masked, relPath, rawLines) {
|
|
|
233
273
|
let matched = null
|
|
234
274
|
let joinedText = null
|
|
235
275
|
|
|
236
|
-
//
|
|
237
|
-
const
|
|
238
|
-
if (
|
|
239
|
-
const
|
|
240
|
-
if (
|
|
241
|
-
|
|
242
|
-
}
|
|
276
|
+
// interface / type(支持 export 与多行;没有 TS 增强器时这是唯一来源)
|
|
277
|
+
const typeDecl = readTypeDeclaration(rawLines, i)
|
|
278
|
+
if (typeDecl) {
|
|
279
|
+
const sig = formatTypeDeclaration(typeDecl)
|
|
280
|
+
if (sig) matched = { name: typeDecl.name, kind: typeDecl.kind, typeSig: sig }
|
|
281
|
+
i = typeDecl.endIdx
|
|
243
282
|
}
|
|
244
283
|
|
|
245
284
|
if (!matched) {
|
|
@@ -294,7 +333,7 @@ function scanJsLike(masked, relPath, rawLines) {
|
|
|
294
333
|
if (overloads.length > 1) matched.overloads = overloads
|
|
295
334
|
}
|
|
296
335
|
|
|
297
|
-
symbols.push(buildSymbol(matched, relPath, rawLines[
|
|
336
|
+
symbols.push(buildSymbol(matched, relPath, rawLines[declStart], declStart + 1))
|
|
298
337
|
}
|
|
299
338
|
|
|
300
339
|
prevOpensBlock = masked[i].trim().endsWith('{')
|
|
@@ -433,7 +472,7 @@ function buildSymbol(matched, relPath, rawLine, lineNo) {
|
|
|
433
472
|
export function scanSymbols(a, b, c) {
|
|
434
473
|
// Backward compatible: old signature (filePath, content) or new (relPath, filePath, content)
|
|
435
474
|
const [relPath, filePath, content] = c === undefined ? [a, a, b] : [a, b, c]
|
|
436
|
-
const ext = relPath.slice(relPath.lastIndexOf('.'))
|
|
475
|
+
const ext = relPath.slice(relPath.lastIndexOf('.')).toLowerCase()
|
|
437
476
|
const lines = content.split(/\r?\n/)
|
|
438
477
|
if (JS_LIKE.has(ext)) return scanJsLike(maskTokens(lines, JS_MASKER), relPath, lines)
|
|
439
478
|
if (PYTHON.has(ext)) return scanPython(maskTokens(lines, PY_MASKER), relPath, lines)
|
package/src/tools/forget.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { defineTool } from '@deepseek-ai/dsh-tools'
|
|
2
|
-
import { memoryRootFor, resolveIndexRoot } from '../util/fs.js'
|
|
2
|
+
import { assertIndexRoot, memoryRootFor, resolveIndexRoot } from '../util/fs.js'
|
|
3
3
|
import { ProjectMemoryStore } from '../store.js'
|
|
4
4
|
|
|
5
5
|
export function forgetTool(config) {
|
|
@@ -24,6 +24,7 @@ export function forgetTool(config) {
|
|
|
24
24
|
},
|
|
25
25
|
async execute(args, exec) {
|
|
26
26
|
const root = resolveIndexRoot(exec, args.root)
|
|
27
|
+
assertIndexRoot(root, args.root || root)
|
|
27
28
|
const memoryDir = memoryRootFor(root, config.memoryDir)
|
|
28
29
|
const store = new ProjectMemoryStore(memoryDir).load()
|
|
29
30
|
return store.commit((s) => {
|
package/src/tools/index-doc.js
CHANGED
|
@@ -2,6 +2,7 @@ import { defineTool } from '@deepseek-ai/dsh-tools'
|
|
|
2
2
|
import path from 'node:path'
|
|
3
3
|
import { assertReadableFile, memoryRootFor, sha256OfFile, storeKey } from '../util/fs.js'
|
|
4
4
|
import { buildDocEntries } from '../doc-pipeline.js'
|
|
5
|
+
import { docEntriesNeedBackfill } from '../doc-index.js'
|
|
5
6
|
import { linkEntries } from '../link.js'
|
|
6
7
|
import { ProjectMemoryStore } from '../store.js'
|
|
7
8
|
import { findProjectRoot } from '../lazy.js'
|
|
@@ -38,7 +39,9 @@ export function indexDocTool(ctx, config) {
|
|
|
38
39
|
const rel = storeKey(path.relative(root, filePath).split(path.sep).join('/'))
|
|
39
40
|
const { hash, size } = await sha256OfFile(filePath)
|
|
40
41
|
const existing = store.fileRecord(rel)
|
|
41
|
-
|
|
42
|
+
// 与 index_repo/watch/lazy 同一条判据:哈希未变但旧条目缺 terms 时仍要重抽一次(一次性回填)。
|
|
43
|
+
// 少了这一步,terms 回填就是"路径相关"的——只有走 index_repo 才生效。
|
|
44
|
+
if (existing && existing.sha256 === hash && !docEntriesNeedBackfill(store.entries[rel])) {
|
|
42
45
|
return `Skipped (unchanged): ${rel}\nAlready indexed with ${(store.entries[rel] || []).length} entry/entries.`
|
|
43
46
|
}
|
|
44
47
|
|