@yolk_vat-y/dsh-project-memory 0.5.6 → 0.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/lazy.js CHANGED
@@ -1,11 +1,8 @@
1
1
  import path from 'node:path'
2
2
  import { tmpdir } from 'node:os'
3
3
  import { existsSync, readdirSync, statSync } from 'node:fs'
4
- import { isSupportedCode, isSupportedDoc, memoryRootFor, readFileForIndex, relativePath, storeKey } from './util/fs.js'
5
- import { buildDocEntries } from './doc-pipeline.js'
6
- import { docEntriesNeedBackfill } from './doc-index.js'
7
- import { scanSymbols } from './symbols.js'
8
- import { linkEntries } from './link.js'
4
+ import { memoryRootFor, relativePath, storeKey } from './util/fs.js'
5
+ import { DROP, OVERSIZE, UNCHANGED, commitFileUpdates, fileKind, planFileIndex, toFileUpdate } from './index-pipeline.js'
9
6
  import { ProjectMemoryStore } from './store.js'
10
7
  import { onFileObserved } from './enhancer.js'
11
8
 
@@ -78,70 +75,42 @@ export function findProjectRoot(filePath, ceiling = path.resolve(tmpdir())) {
78
75
  }
79
76
 
80
77
  export async function indexFile(ctx, config, filePath, watchManager = null) {
81
- const ext = path.extname(filePath).toLowerCase()
82
- if (!isSupportedDoc(ext) && !isSupportedCode(ext)) return false
78
+ const kind = fileKind(path.extname(filePath).toLowerCase())
79
+ if (!kind) return false
83
80
  const root = findProjectRoot(filePath)
84
81
  if (!root) return false
85
82
 
86
83
  const memoryDir = memoryRootFor(root, config.memoryDir)
87
84
  const store = new ProjectMemoryStore(memoryDir).load()
88
85
  const rel = storeKey(relativePath(root, filePath))
89
- const existing = store.fileRecord(rel)
90
- let hash
86
+ const record = store.fileRecord(rel)
87
+
91
88
  let size
92
- let buffer
93
- let entries
89
+ let plan
94
90
  try {
95
- if (isSupportedCode(ext) && config.maxFileSizeMb && statSync(filePath).size > config.maxFileSizeMb * 1024 * 1024) {
96
- return false
97
- }
98
- ;({ hash, size, buffer } = readFileForIndex(filePath))
91
+ size = statSync(filePath).size
92
+ plan = await planFileIndex({ rel, filePath, kind, config, record, existingEntries: store.entries[rel], size })
99
93
  } catch {
100
94
  return false
101
95
  }
102
- // 旧 store 的 doc 条目缺 terms → 一次性回填(即使哈希未变)
103
- if (existing && existing.sha256 === hash && !(isSupportedDoc(ext) && docEntriesNeedBackfill(store.entries[rel]))) return false
96
+ if (plan.key === UNCHANGED || plan.key === OVERSIZE) return false
104
97
 
105
98
  if (watchManager) {
106
99
  watchManager.addRoot(root)
107
100
  store.addWatch(root)
108
101
  }
109
102
 
110
- if (isSupportedCode(ext)) {
111
- entries = scanSymbols(rel, filePath, buffer.toString('utf8'))
103
+ if (plan.key !== DROP && plan.type === 'code') {
112
104
  onFileObserved(store, rel, filePath, config, root)
113
- return store.commit((s) => {
114
- s.markFile(rel, { sha256: hash, size, type: 'code', indexedAt: new Date().toISOString() })
115
- s.setEntries(rel, entries)
116
- linkEntries(s)
117
- return true
118
- })
119
- } else {
120
- entries = await buildDocEntries(rel, filePath, {
121
- chunkChars: config.chunkChars,
122
- maxChunks: config.maxChunksPerFile,
123
- maxFileSizeMb: config.maxFileSizeMb,
124
- maxPdfPages: config.maxPdfPages,
125
- })
126
- if (entries === null) {
127
- return store.commit((s) => {
128
- s.removeFile(rel)
129
- return false
130
- })
131
- }
132
- return store.commit((s) => {
133
- s.markFile(rel, { sha256: hash, size, type: 'doc', indexedAt: new Date().toISOString() })
134
- s.setEntries(rel, entries)
135
- linkEntries(s)
136
- return true
137
- })
138
105
  }
106
+ commitFileUpdates(store, { updates: [toFileUpdate(rel, plan, record)] })
107
+ return plan.key !== DROP
139
108
  }
140
109
 
141
110
  export function codeFirst(paths) {
142
111
  return [...paths].sort((a, b) => {
143
- const aCode = isSupportedCode(path.extname(a).toLowerCase()) ? 0 : 1
144
- const bCode = isSupportedCode(path.extname(b).toLowerCase()) ? 0 : 1
112
+ const aCode = fileKind(path.extname(a).toLowerCase()) === 'code' ? 0 : 1
113
+ const bCode = fileKind(path.extname(b).toLowerCase()) === 'code' ? 0 : 1
145
114
  return aCode - bCode
146
115
  })
147
116
  }
package/src/link.js CHANGED
@@ -39,7 +39,16 @@ export function linkEntries(store) {
39
39
  if (linked.size > before) links++
40
40
  }
41
41
  }
42
- doc.linkedSymbols = linked.size ? [...linked] : undefined
42
+ const before = Array.isArray(doc.linkedSymbols) ? doc.linkedSymbols.join('\u0000') : ''
43
+ const next = [...linked]
44
+ if (before === next.join('\u0000')) continue
45
+ doc.linkedSymbols = next.length ? next : undefined
46
+ // 链接是在**符号**落盘那一刻算出来的,此时 doc 的 shard 往往不是脏的;不标脏就只存在于内存,
47
+ // 下次进程启动重新加载后链接全部丢失("文档先索引、符号后到"的正常顺序)。
48
+ if (typeof store.markFile === 'function' && typeof store.fileRecord === 'function' && doc.sourcePath) {
49
+ const record = store.fileRecord(doc.sourcePath)
50
+ if (record) store.markFile(doc.sourcePath, record)
51
+ }
43
52
  }
44
53
  return links
45
54
  }
@@ -60,9 +60,8 @@ export async function parsePdf(filePath, { pages = null, maxPages = 1000, backen
60
60
  const data = new Uint8Array(await readFile(filePath))
61
61
  const { getDocument } = await loadPdfjs()
62
62
  const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
63
- const doc = await loadingTask.promise
64
-
65
63
  try {
64
+ const doc = await loadingTask.promise
66
65
  const total = doc.numPages
67
66
  if (total > maxPages) {
68
67
  throw new Error(`PDF has ${total} pages, over the maxPages limit of ${maxPages}`)
@@ -112,9 +111,8 @@ export async function parsePdfInfo(filePath, maxPages = 1000) {
112
111
  const data = new Uint8Array(await readFile(filePath))
113
112
  const { getDocument } = await loadPdfjs()
114
113
  const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
115
- const doc = await loadingTask.promise
116
-
117
114
  try {
115
+ const doc = await loadingTask.promise
118
116
  if (doc.numPages > maxPages) {
119
117
  throw new Error(`PDF has ${doc.numPages} pages, over the maxPages limit of ${maxPages}`)
120
118
  }
@@ -17,11 +17,18 @@ function depsOf(manifest, fields) {
17
17
  const tags = new Set()
18
18
  for (const f of fields) {
19
19
  const deps = manifest && manifest[f]
20
- if (deps && typeof deps === 'object') {
21
- for (const name of Object.keys(deps)) {
22
- tags.add(name.toLowerCase())
23
- if (name.startsWith('@')) tags.add(name.split('/')[1].toLowerCase())
24
- }
20
+ if (!deps || typeof deps !== 'object') continue
21
+ for (const name of Object.keys(deps)) {
22
+ if (typeof name !== 'string' || !name) continue
23
+ tags.add(name.toLowerCase())
24
+ if (!name.startsWith('@')) continue
25
+ // scoped 包名 `@scope/pkg`:要加的是 **scope**(`vue`),不是包名段(`compiler-sfc`)。
26
+ // 旧写法 `name.split('/')[1]` 让 `@vue/*` 项目永远拿不到 `vue` tag,
27
+ // trigger.scope:['vue'] / legacy scope 过滤会把最相关的那条 procedure 静默判死;
28
+ // 而 `@malformed`(没有 `/`)会在这里抛错,异常被 collectTags 整个吞掉 → 全项目 tags 清零。
29
+ const parts = name.slice(1).split('/')
30
+ if (parts[0]) tags.add(parts[0].toLowerCase())
31
+ if (parts[1]) tags.add(parts[1].toLowerCase())
25
32
  }
26
33
  }
27
34
  return tags
package/src/readiness.js CHANGED
@@ -82,7 +82,16 @@ export function buildReadinessContext(input = {}) {
82
82
  const humanText = String(input.humanText ?? input.query ?? '')
83
83
  const actionText = String(input.actionText || '')
84
84
  const actions = new Set([...(input.actions || []), ...detectActions(humanText), ...detectActions(actionText)])
85
- const paths = new Set([...(input.paths || []), ...extractPaths(actionText), ...extractPaths(humanText)])
85
+ // 人类消息里提到的路径 = **动手前的写意图**("帮我改 src/x.js");动作文本里的路径不区分读写,
86
+ // 读一个文件不该算写目标(否则 `read README.md` 会命中 writes:['README.md'])。
87
+ // 两者对外仍合并成 `paths`(兼容既有语义),写通道单看 humanPaths。
88
+ // 调用方若已带 humanPaths(injectForStep 构建后再传给 buildInjection),以它为准,
89
+ // 否则退回 paths —— 避免把上一次算出来的并集(含读路径)再当写意图。
90
+ const humanPaths = new Set([
91
+ ...(Array.isArray(input.humanPaths) ? input.humanPaths : (input.paths || [])),
92
+ ...extractPaths(humanText),
93
+ ])
94
+ const paths = new Set([...humanPaths, ...extractPaths(actionText)])
86
95
  // 动作平面(S1):结构化调用优先;没有结构化调用时对 actionText 跑一遍 shell 规则兜底。
87
96
  const fromText = activityFromText(actionText)
88
97
  const ops = new Set([...(input.ops || []), ...fromText.ops])
@@ -97,6 +106,7 @@ export function buildReadinessContext(input = {}) {
97
106
  actionText,
98
107
  actions: [...actions],
99
108
  paths: [...paths],
109
+ humanPaths: [...humanPaths],
100
110
  ops: [...ops],
101
111
  targets: [...targets],
102
112
  hosts: [...hosts],
@@ -209,6 +219,10 @@ export function intentText(text) {
209
219
  .replace(/“[^”\n]*”/g, ' ')
210
220
  .replace(/[A-Za-z]:\\[^\s"']*/g, ' ')
211
221
  .replace(/(?:[A-Za-z0-9_.@-]+\/)+[A-Za-z0-9_.@-]+/g, ' ')
222
+ // 带扩展名的**非 ASCII 文件名**(`石啸天-记忆方向调研.pptx` / `docs/面试演示-王金鹏.md`):
223
+ // 只挡 ASCII 路径不够——中文文件名整块留下,"调研""面试"这类子串照样触发 when.intents,
224
+ // 正是引号剥离要防的那类假阳性。ASCII 裸名(如 de-TODO.md)保持原语义,不在这里动。
225
+ .replace(/\S*[^\x00-\x7F]\S*\.[A-Za-z][A-Za-z0-9]{0,7}\b/g, ' ')
212
226
  // 仓库里的裸文件名(README / CHANGELOG …)也是语料,不是意图
213
227
  .replace(/\b(?:README|CHANGELOG|LICENSE|AGENTS|CONTRIBUTING|Dockerfile|Makefile)\b/g, ' ')
214
228
  .replace(/\s+/g, ' ')
@@ -301,10 +315,14 @@ export function matchTrigger(trigger, ctx) {
301
315
  for (const op of when.ops || []) {
302
316
  if (op && (ctx?.ops || []).includes(String(op))) return `op:${op}`
303
317
  }
318
+ // 写目标 = 已观察到的结构化写调用 ∪ 人类消息里提到的路径(动手前的写意图)。
319
+ // DSH 没有工具执行前拦截钩子,动手前唯一能拿到的写意图就是人类消息里的路径;
320
+ // 但动作文本里的**读**路径不算(`read README.md` 不是"即将写 README.md")。
321
+ const writeTargets = [...(ctx?.targets || []), ...(ctx?.humanPaths || [])]
304
322
  for (const w of when.writes || []) {
305
323
  // 扩展名/泛名 glob 在这里被硬性忽略:`*.pptx` 这类条件只能撒谎,不能收窄。
306
324
  if (!w || isDroppableGlob(w)) continue
307
- if (matchAnyPath(w, ctx?.targets)) return `write:${w}`
325
+ if (matchAnyPath(w, writeTargets)) return `write:${w}`
308
326
  }
309
327
  const intent = ctx?.intent ?? ctx?.humanText ?? ''
310
328
  for (const it of when.intents || []) {
@@ -355,6 +373,9 @@ export function normalizeTrigger(it, opts = {}) {
355
373
  // (实测:那样会让 D 场景一次多出 6 条假阳性)。
356
374
  const intents = []
357
375
  for (const k of t.keywords || []) if (isIntentWord(k)) intents.push(String(k))
376
+ // 旧 `symbols` 也走文本平面:不归一就等于静默丢掉一个作者写下的触发面
377
+ // (README 承诺 keywords/symbols/actions/paths/scope 都会迁移)。与 keywords 同一把准入尺子。
378
+ for (const s of t.symbols || []) if (isIntentWord(s)) intents.push(String(s))
358
379
  const when = {}
359
380
  if (ops.size) when.ops = [...ops].sort()
360
381
  if (writes.length) when.writes = [...new Set(writes)]
@@ -463,6 +484,9 @@ export function relativeHits(scored, { ratioMin = 0.5 } = {}) {
463
484
  const list = (scored || []).filter((r) => r && Number.isFinite(r.score) && r.score > 0)
464
485
  if (!list.length) return []
465
486
  const top = list[0].score
466
- const floor = top * (typeof ratioMin === 'number' && ratioMin > 0 ? ratioMin : 0.5)
487
+ // 显式的 0 表示"关掉相对门槛"(只剩 score>0);未给/非法值才回退 0.5。
488
+ // 旧写法 `ratioMin > 0 ? ratioMin : 0.5` 把 0 当成"没配",配置上无法关闭。
489
+ const ratio = typeof ratioMin === 'number' && Number.isFinite(ratioMin) ? ratioMin : 0.5
490
+ const floor = top * Math.max(0, Math.min(1, ratio))
467
491
  return list.filter((r) => r.score >= floor)
468
492
  }
package/src/recall.js CHANGED
@@ -101,6 +101,21 @@ export function insightToEntry(ins) {
101
101
  }
102
102
  }
103
103
 
104
+ /**
105
+ * 一条 insight 的**打分文本**(BM25 与 IDF 覆盖率必须用同一份)。
106
+ *
107
+ * 为什么要有这个函数:提示通道的"这条查询离语料有多远"(`idfCoverage` 的 df)曾经用
108
+ * `insightMatchText` 计算,而排序用 `insightToEntry`(含 fix/solution/reason/steps)。
109
+ * 于是"只在解药里有匹配"的查询词 df=0 → supportRatio=0 → **整条提示通道本轮沉默**,
110
+ * 而排序明明给了它高分。两份文本永远来自同一个 entry,才不会再漂移。
111
+ * @param {object} ins
112
+ * @returns {string}
113
+ */
114
+ export function insightScoringText(ins) {
115
+ const e = insightToEntry(ins)
116
+ return [e.title, e.summary, e.terms, ...(e.keywords || [])].filter(Boolean).join('\n')
117
+ }
118
+
104
119
  /**
105
120
  * 召回可见的 insight 集合:作用域可见性跟随会话绑定。
106
121
  * - project / global:始终可见(归档除外);
@@ -127,6 +142,9 @@ export function visibleInsights({ store = null, globalStore = null, boundTaskId
127
142
  const task = store.getTask(boundTaskId)
128
143
  for (const it of (task && task.insights) || []) {
129
144
  if (!it || it.archived) continue
145
+ // 任务级也要挡草稿:reflection 写的是 draft:true,未确认前不该被 query_memory 召回
146
+ // (project/global 一直有这道闸,任务级漏了)。
147
+ if (it.draft === true) continue
130
148
  out.push({ ...it, scope: it.scope || 'task' })
131
149
  }
132
150
  }
@@ -163,15 +181,17 @@ export function recallItems(opts = {}) {
163
181
  if (want.has('doc') || want.has('symbol')) {
164
182
  const pool = (entries || (store && typeof store.allEntries === 'function' ? store.allEntries() : []) || [])
165
183
  .filter((e) => e && (e.type === 'doc' || e.type === 'symbol') && want.has(e.type))
166
- const scored = rankEntriesStreaming(pool, qs, idf, limit)
167
- const byLayer = new Map([['doc', []], ['symbol', []]])
168
- for (const { entry, score } of scored) {
169
- const prior = entry.type === 'doc' ? normativePrior(entry) : LAYER_PRIORS.symbol
170
- byLayer.get(entry.type).push({ item: entry, score, prior, weightedScore: score * prior })
171
- }
184
+ // 每层各自 top-k(不变量 #2):一次跨类型取 top-k 会让文档把符号挤出结果——
185
+ // 20 条命中文档 + 1 条命中符号、limit 8 时符号层直接消失。分类型各取一次再分桶。
172
186
  for (const layer of ['doc', 'symbol']) {
173
- const hits = byLayer.get(layer)
174
- if (!hits.length) continue
187
+ const layerPool = pool.filter((e) => e.type === layer)
188
+ if (!layerPool.length) continue
189
+ const scored = rankEntriesStreaming(layerPool, qs, idf, limit)
190
+ if (!scored.length) continue
191
+ const hits = scored.map(({ entry, score }) => {
192
+ const prior = layer === 'doc' ? normativePrior(entry) : LAYER_PRIORS.symbol
193
+ return { item: entry, score, prior, weightedScore: score * prior }
194
+ })
175
195
  hits.sort((a, b) => b.weightedScore - a.weightedScore)
176
196
  const top = hits[0].weightedScore || 1
177
197
  buckets.push({
@@ -2,7 +2,7 @@
2
2
  // 事实依据:dsh session/event 签名 (session, event);todo/write data.todos;
3
3
  // tool/call data.arguments 为 JSON 字符串;fs 工具名 read/write/edit/read_image,参数 file_path。
4
4
  import path from 'node:path'
5
- import { createHash } from 'node:crypto'
5
+ import { createHash, randomUUID } from 'node:crypto'
6
6
  import { memoryRootFor } from '../util/fs.js'
7
7
  import { findProjectRoot } from '../lazy.js'
8
8
  import { ProjectMemoryStore } from '../store.js'
@@ -27,7 +27,9 @@ export function slugifyTitle(text) {
27
27
 
28
28
  export function genTaskId(projectRoot, title) {
29
29
  const slug = slugifyTitle(title) || 'task'
30
- return `tsk_${hash8(projectRoot)}_${slug}_${Date.now()}`
30
+ // 毫秒不是唯一性保证:两个会话在同一毫秒建同名任务会拿到同一个 id,
31
+ // addTask 变成"同一 id 两次",两个会话的绑定互相串台。补 6 位随机后缀。
32
+ return `tsk_${hash8(projectRoot)}_${slug}_${Date.now().toString(36)}_${randomUUID().slice(0, 6)}`
31
33
  }
32
34
 
33
35
  /** 项目根推导:findProjectRoot 期望文件路径,传目录会从父级起跳,故用目录内探针路径。 */
@@ -138,13 +140,7 @@ export function hotSortFiles(files, meta) {
138
140
  /** 触碰文件后更新元数据并重排 task.files(写 vs 读分别记时间/次数)。 */
139
141
  export function touchTaskFile(task, rel, kind, now) {
140
142
  task.fileMeta = task.fileMeta || {}
141
- if (!task.files.includes(rel)) {
142
- task.files.push(rel)
143
- if (task.files.length > MAX_FILES_PER_TASK) {
144
- const dropped = task.files.shift()
145
- delete task.fileMeta[dropped]
146
- }
147
- }
143
+ if (!task.files.includes(rel)) task.files.push(rel)
148
144
  const m = task.fileMeta[rel] || {}
149
145
  m.n = (m.n || 0) + 1 // 兼容旧字段:读+写总数(保留,避免老任务语义漂移)
150
146
  // 读/写分计数(先只采集,不消费 —— 排序与注入仍只用 lastWriteAt/lastReadAt)
@@ -157,6 +153,12 @@ export function touchTaskFile(task, rel, kind, now) {
157
153
  m.lastReadAt = now
158
154
  task.fileMeta[rel] = m
159
155
  task.files = hotSortFiles(task.files, task.fileMeta)
156
+ // 超限丢最冷的:列表已按热度降序,末尾即最冷。旧代码在**重排前** shift(),
157
+ // 而那时 files 本身就是热序,shift 恰好把最热的那个文件(刚写的)丢掉、留下最冷的。
158
+ while (task.files.length > MAX_FILES_PER_TASK) {
159
+ const dropped = task.files.pop()
160
+ delete task.fileMeta[dropped]
161
+ }
160
162
  }
161
163
 
162
164
  /**
package/src/store.js CHANGED
@@ -17,6 +17,14 @@ const SHARDS_DIR = 'shards'
17
17
  const storeCache = new Map()
18
18
  const STORE_CACHE_MAX = 32
19
19
 
20
+ /** 已就"无法迁移的旧 store"告警过的目录:避免每次 load() 都刷一行。 */
21
+ const migrationWarned = new Set()
22
+
23
+ /** 纯对象判定(排除 null / 数组):磁盘读入的 JSON 形状校验统一走它。 */
24
+ function isRecord(value) {
25
+ return Boolean(value) && typeof value === 'object' && !Array.isArray(value)
26
+ }
27
+
20
28
  function loadJson(filePath, fallback) {
21
29
  let raw
22
30
  try {
@@ -82,7 +90,9 @@ export class ProjectMemoryStore {
82
90
  load() {
83
91
  const key = path.resolve(this.dir)
84
92
  const hot = storeCache.get(key)
85
- if (hot && hot !== this) return hot
93
+ // 缓存命中即返回:`hot === this` 时再读一遍盘会静默丢弃本实例尚未 save() 的变更
94
+ // (_loadSharded/_loadInsights 会重新赋值 experience/tasks/insights…)。
95
+ if (hot) return hot
86
96
  this._migrateLegacyIfNeeded()
87
97
  this._loadSharded()
88
98
  this._loadInsights()
@@ -114,9 +124,34 @@ export class ProjectMemoryStore {
114
124
  }
115
125
  const legacyEntriesPath = path.join(this.dir, ENTRIES_FILE)
116
126
  if (!existsSafe(legacyEntriesPath)) return
117
- const index = loadJson(path.join(this.dir, INDEX_FILE), {})
118
- const files = index.files || {}
119
- const entries = loadJson(legacyEntriesPath, {})
127
+ const legacyIndexPath = path.join(this.dir, INDEX_FILE)
128
+ // 先读两份旧文件:损坏的那份会在这里被 loadJson 备份为 .corrupt(保留现场)。
129
+ const index = loadJson(legacyIndexPath, null)
130
+ const entries = loadJson(legacyEntriesPath, null)
131
+ if (!isRecord(index) || !isRecord(entries)) {
132
+ const hasEntries = isRecord(entries) && Object.keys(entries).length > 0
133
+ if (!hasEntries) {
134
+ // entries 损坏(已备份)或本就是空的:没有可保护的数据,按空旧库收尾。
135
+ writeJsonAtomic(formatPath, { version: 2, layout: 'sharded' })
136
+ for (const stale of [legacyEntriesPath, legacyIndexPath]) {
137
+ try {
138
+ unlinkSync(stale)
139
+ } catch {
140
+ // already renamed away by corrupt backup, or gone; nothing to do
141
+ }
142
+ }
143
+ return
144
+ }
145
+ // index.json 是 entries.json → rel 的唯一映射。它缺失或损坏时继续迁移,会写出 0 个
146
+ // shard、打上 v2 标记、再把**完好的** entries.json 删掉——等于一次静默的数据清空。
147
+ // 保留现场,不写 format 标记,等 index.json 修好后再迁(每次进程只提示一次)。
148
+ if (!migrationWarned.has(this.dir)) {
149
+ migrationWarned.add(this.dir)
150
+ console.error(`[dsh-project-memory] legacy store at ${this.dir} has ${ENTRIES_FILE} but no readable ${INDEX_FILE}; migration skipped to protect it`)
151
+ }
152
+ return
153
+ }
154
+ const files = isRecord(index.files) ? index.files : {}
120
155
  const orphans = Object.keys(entries).filter((rel) => !(rel in files))
121
156
  if (orphans.length) {
122
157
  console.error(
@@ -147,9 +182,11 @@ export class ProjectMemoryStore {
147
182
  }
148
183
  for (const name of shardNames) {
149
184
  const shard = loadJson(path.join(this.dir, SHARDS_DIR, name), null)
150
- if (!shard || typeof shard.relPath !== 'string' || !shard.record) continue
185
+ if (!shard || typeof shard.relPath !== 'string' || !isRecord(shard.record)) continue
151
186
  this.files[shard.relPath] = shard.record
152
- this.entries[shard.relPath] = shard.entries || []
187
+ // 畸形 shard(entries 被写成对象/null)不能让 allEntries() 在 `for…of` 上抛错,
188
+ // 否则一个坏文件会拖垮整个进程的每一次读取。
189
+ this.entries[shard.relPath] = Array.isArray(shard.entries) ? shard.entries.filter(isRecord) : []
153
190
  }
154
191
  this.experience = loadJson(path.join(this.dir, EXPERIENCE_FILE), [])
155
192
  this.tasks = loadJson(path.join(this.dir, TASKS_FILE), [])
@@ -165,7 +202,10 @@ export class ProjectMemoryStore {
165
202
  // migratedAt 落盘保证跨进程/崩溃幂等。销毁式收敛放到 recall 统一 PR。
166
203
  _loadInsights() {
167
204
  const doc = loadJson(path.join(this.dir, INSIGHTS_FILE), null)
168
- this.insights = doc && typeof doc === 'object' && Array.isArray(doc.items) ? doc : { version: 1, migratedAt: null, items: [] }
205
+ this.insights = isRecord(doc) && Array.isArray(doc.items)
206
+ // items 里混进 null/非对象(手改或旧版写入)会让迁移与召回逐个 `.title` 抛错——过滤掉。
207
+ ? { ...doc, items: doc.items.filter(isRecord) }
208
+ : { version: 1, migratedAt: null, items: [] }
169
209
  this._migrateExperienceToInsights()
170
210
  // PR3:v1 → v2 懒回填派生 trigger(纯确定性、幂等;不调用模型,不改写已有字段)
171
211
  if (backfillDerivedTriggers(this.insights)) this._dirtyInsights = true
@@ -551,15 +591,14 @@ export class ProjectMemoryStore {
551
591
  return result
552
592
  }
553
593
 
554
- applyFileUpdate(relPath, { expectedHash, hash, entries, meta, deleted, type, size }) {
555
- const cur = this.fileRecord(relPath)
556
- if (!deleted && (cur?.sha256 ?? null) !== (expectedHash ?? null)) {
557
- return { skipped: true }
558
- }
559
- if (deleted) {
560
- this.removeFile(relPath)
561
- return { ok: true }
562
- }
594
+ /**
595
+ * 写入一条文件更新,以 expectedHash 做 CAS。
596
+ * @returns {boolean} true = 已写入;false = 文件自扫描后又被改动,本次拒绝
597
+ * (调用方让该条目保持未落快照,下一轮重试)。
598
+ * 移除条目不经过这里:调用方在 commit 里直接 removeFile(见 index-pipeline.js)。
599
+ */
600
+ applyFileUpdate(relPath, { expectedHash, hash, entries, type, size }) {
601
+ if ((this.fileRecord(relPath)?.sha256 ?? null) !== (expectedHash ?? null)) return false
563
602
  this.markFile(relPath, {
564
603
  sha256: hash,
565
604
  size,
@@ -567,7 +606,7 @@ export class ProjectMemoryStore {
567
606
  indexedAt: new Date().toISOString(),
568
607
  })
569
608
  this.setEntries(relPath, entries)
570
- return { ok: true }
609
+ return true
571
610
  }
572
611
  }
573
612
 
package/src/symbols.js CHANGED
@@ -144,18 +144,57 @@ function extractTypeSignature(line) {
144
144
  return sig
145
145
  }
146
146
 
147
- function extractInterfaceOrType(line) {
148
- // interface User { name: string; age: number }
149
- const ifaceMatch = line.match(/^interface\s+(\w+)\s*(?:extends\s+[^{]+)?\s*\{([^}]*)\}/)
150
- if (ifaceMatch) {
151
- return `{ ${ifaceMatch[2].trim()} }`
147
+ function braceDelta(s) {
148
+ let d = 0
149
+ for (const ch of s) {
150
+ if (ch === '{') d++
151
+ else if (ch === '}') d--
152
152
  }
153
- // type UserMap = Map<string, User>
154
- const typeMatch = line.match(/^type\s+(\w+)\s*=\s*([^;{]+)/)
155
- if (typeMatch) {
156
- return `= ${typeMatch[2].trim()}`
153
+ return d
154
+ }
155
+
156
+ /**
157
+ * 读取一条 interface / type 声明,支持 `export`(含 `declare`)与多行形态。
158
+ * 旧实现要求 `^interface`(不吃 export)且 `}` 必须在本行,于是
159
+ * `export interface X {…}`、任何多行 interface、`export type X = …` 在 L1 正则扫描器里
160
+ * 全都产出 0 个符号(TS 增强器可用时才被补回来;没有 typescript 的项目就彻底看不到)。
161
+ * @returns {{kind: 'interface'|'type', name: string, text: string, endIdx: number} | null}
162
+ */
163
+ function readTypeDeclaration(rawLines, startIdx) {
164
+ const first = rawLines[startIdx].trim()
165
+ const m = first.match(/^(?:export\s+)?(?:declare\s+)?(interface|type)\s+([A-Za-z_$][\w$]*)/)
166
+ if (!m) return null
167
+ const kind = m[1]
168
+ const parts = [first]
169
+ let depth = braceDelta(first)
170
+ let endIdx = startIdx
171
+ const complete = () => (kind === 'interface'
172
+ ? depth <= 0 && parts[parts.length - 1].includes('}')
173
+ : depth <= 0 && /[;}]/.test(parts[parts.length - 1]))
174
+ for (let j = startIdx + 1; j < rawLines.length && j - startIdx <= 20 && !complete(); j++) {
175
+ const t = rawLines[j].trim()
176
+ if (!t) {
177
+ if (depth <= 0) break
178
+ parts.push(t)
179
+ endIdx = j
180
+ continue
181
+ }
182
+ // 归零后遇到下一条声明开头就收尾(TS 的 type 别名常不写分号)
183
+ if (depth <= 0 && j > startIdx && /^(?:export\s+)?(?:declare\s+)?(?:interface|type|function|class|const|let|var|import|enum)\b/.test(t)) break
184
+ parts.push(t)
185
+ depth += braceDelta(t)
186
+ endIdx = j
157
187
  }
158
- return ''
188
+ return { kind, name: m[2], text: parts.join(' '), endIdx }
189
+ }
190
+
191
+ function formatTypeDeclaration({ kind, text }) {
192
+ if (kind === 'interface') {
193
+ const body = text.match(/\{([\s\S]*)\}/)
194
+ return body ? `{ ${body[1].replace(/\s+/g, ' ').trim()} }` : ''
195
+ }
196
+ const eq = text.match(/=\s*([\s\S]+?);?\s*$/)
197
+ return eq ? `= ${eq[1].replace(/\s+/g, ' ').trim()}` : ''
159
198
  }
160
199
 
161
200
  function extractOverloads(masked, startIdx) {
@@ -226,6 +265,7 @@ function scanJsLike(masked, relPath, rawLines) {
226
265
  const symbols = []
227
266
  let prevOpensBlock = false
228
267
  for (let i = 0; i < masked.length; i++) {
268
+ const declStart = i // 多行声明会推进 i,符号行号要记声明的**首行**
229
269
  const rawText = rawLines[i].trim()
230
270
  const maskedText = masked[i].trim()
231
271
  if (!rawText) continue
@@ -233,13 +273,12 @@ function scanJsLike(masked, relPath, rawLines) {
233
273
  let matched = null
234
274
  let joinedText = null
235
275
 
236
- // Check for interface / type alias first (on raw line, not masked)
237
- const ifaceSig = extractInterfaceOrType(rawText)
238
- if (ifaceSig) {
239
- const nameMatch = rawText.match(/^(?:export\s+)?(?:interface|type)\s+(\w+)/)
240
- if (nameMatch) {
241
- matched = { name: nameMatch[1], kind: 'interface', typeSig: ifaceSig }
242
- }
276
+ // interface / type(支持 export 与多行;没有 TS 增强器时这是唯一来源)
277
+ const typeDecl = readTypeDeclaration(rawLines, i)
278
+ if (typeDecl) {
279
+ const sig = formatTypeDeclaration(typeDecl)
280
+ if (sig) matched = { name: typeDecl.name, kind: typeDecl.kind, typeSig: sig }
281
+ i = typeDecl.endIdx
243
282
  }
244
283
 
245
284
  if (!matched) {
@@ -294,7 +333,7 @@ function scanJsLike(masked, relPath, rawLines) {
294
333
  if (overloads.length > 1) matched.overloads = overloads
295
334
  }
296
335
 
297
- symbols.push(buildSymbol(matched, relPath, rawLines[i], i + 1))
336
+ symbols.push(buildSymbol(matched, relPath, rawLines[declStart], declStart + 1))
298
337
  }
299
338
 
300
339
  prevOpensBlock = masked[i].trim().endsWith('{')
@@ -433,7 +472,7 @@ function buildSymbol(matched, relPath, rawLine, lineNo) {
433
472
  export function scanSymbols(a, b, c) {
434
473
  // Backward compatible: old signature (filePath, content) or new (relPath, filePath, content)
435
474
  const [relPath, filePath, content] = c === undefined ? [a, a, b] : [a, b, c]
436
- const ext = relPath.slice(relPath.lastIndexOf('.'))
475
+ const ext = relPath.slice(relPath.lastIndexOf('.')).toLowerCase()
437
476
  const lines = content.split(/\r?\n/)
438
477
  if (JS_LIKE.has(ext)) return scanJsLike(maskTokens(lines, JS_MASKER), relPath, lines)
439
478
  if (PYTHON.has(ext)) return scanPython(maskTokens(lines, PY_MASKER), relPath, lines)
@@ -1,5 +1,5 @@
1
1
  import { defineTool } from '@deepseek-ai/dsh-tools'
2
- import { memoryRootFor, resolveIndexRoot } from '../util/fs.js'
2
+ import { assertIndexRoot, memoryRootFor, resolveIndexRoot } from '../util/fs.js'
3
3
  import { ProjectMemoryStore } from '../store.js'
4
4
 
5
5
  export function forgetTool(config) {
@@ -24,6 +24,7 @@ export function forgetTool(config) {
24
24
  },
25
25
  async execute(args, exec) {
26
26
  const root = resolveIndexRoot(exec, args.root)
27
+ assertIndexRoot(root, args.root || root)
27
28
  const memoryDir = memoryRootFor(root, config.memoryDir)
28
29
  const store = new ProjectMemoryStore(memoryDir).load()
29
30
  return store.commit((s) => {
@@ -2,6 +2,7 @@ import { defineTool } from '@deepseek-ai/dsh-tools'
2
2
  import path from 'node:path'
3
3
  import { assertReadableFile, memoryRootFor, sha256OfFile, storeKey } from '../util/fs.js'
4
4
  import { buildDocEntries } from '../doc-pipeline.js'
5
+ import { docEntriesNeedBackfill } from '../doc-index.js'
5
6
  import { linkEntries } from '../link.js'
6
7
  import { ProjectMemoryStore } from '../store.js'
7
8
  import { findProjectRoot } from '../lazy.js'
@@ -38,7 +39,9 @@ export function indexDocTool(ctx, config) {
38
39
  const rel = storeKey(path.relative(root, filePath).split(path.sep).join('/'))
39
40
  const { hash, size } = await sha256OfFile(filePath)
40
41
  const existing = store.fileRecord(rel)
41
- if (existing && existing.sha256 === hash) {
42
+ // 与 index_repo/watch/lazy 同一条判据:哈希未变但旧条目缺 terms 时仍要重抽一次(一次性回填)。
43
+ // 少了这一步,terms 回填就是"路径相关"的——只有走 index_repo 才生效。
44
+ if (existing && existing.sha256 === hash && !docEntriesNeedBackfill(store.entries[rel])) {
42
45
  return `Skipped (unchanged): ${rel}\nAlready indexed with ${(store.entries[rel] || []).length} entry/entries.`
43
46
  }
44
47