@yolk_vat-y/dsh-project-memory 0.5.9 → 0.5.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/link.js CHANGED
@@ -1,54 +1,120 @@
1
+ // doc → symbol 交叉引用:**读取期解算**,不再物化进 entry。
2
+ //
3
+ // 旧实现(≤0.5.10)在每次索引提交时把「本 chunk 命中的所有符号 id」写进
4
+ // `entry.linkedSymbols` 并落盘。实测一个真实 TypeScript 仓库(本仓库的 361MB store):
5
+ // 17174 个 chunk 共 3,948,420 个链接槽位,只对应 15,456 个符号,单 chunk 最多 2231 个;
6
+ // 而唯一的消费者 `query_memory` 只读前 5 个——存了消费量的 46 倍。
7
+ //
8
+ // 更根本的问题是它把**跨实体派生关系**固化进了源记录:doc 链接的有效性取决于符号表的
9
+ // 当前状态,而符号是独立写入的。于是必须追失效(`markFile` 标脏、"文档先索引、符号后到
10
+ // 则链接丢失"),并且每个索引提交点都要对整库做 O(chunks × symbols) 全表重扫。
11
+ //
12
+ // 现在:entry 只存事实;链接在查询时用每 store 的符号索引解算,成本 O(本 chunk 词数),
13
+ // 且天然反映当前符号表——后索引的符号也能链上,`markFile` 那套失效机制随之删除。
14
+
1
15
  const LATIN_NAME = /^[A-Za-z0-9_]+$/
2
16
  const CJK_CHAR = /[\u3400-\u9fff\uf900-\ufaff\u3040-\u309f\u30a0-\u30ff\uac00-\ud7af]/
17
+ /** 与 latin 边界后视 `[a-z0-9_$]` 同字符集:切出的词直接可查倒排表。 */
18
+ const LATIN_WORD = /[a-z0-9_$]+/g
3
19
 
4
20
  function buildMatcher(name) {
5
21
  const lower = name.toLowerCase()
6
22
  if (LATIN_NAME.test(name)) {
7
- return { lower, re: new RegExp(`(?<![a-z0-9_$])${lower}(?![a-z0-9_$])`) }
23
+ return { re: new RegExp(`(?<![a-z0-9_$])${lower}(?![a-z0-9_$])`) }
8
24
  }
9
25
  const escaped = lower.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
10
26
  // 尾部统一挡 CJK + 字母数字下划线,防止纯 CJK 名误链混合后缀、混合名误链更长后缀
11
- return { lower, re: new RegExp(`${escaped}(?!${CJK_CHAR.source})(?![a-z0-9_$])`) }
27
+ return { re: new RegExp(`${escaped}(?!${CJK_CHAR.source})(?![a-z0-9_$])`) }
12
28
  }
13
29
 
14
- export function linkEntries(store) {
15
- const all = store.allEntries()
16
- const symbols = all.filter((e) => e.type === 'symbol')
17
- const docs = all.filter((e) => e.type === 'doc')
18
- if (!symbols.length || !docs.length) return 0
19
-
20
- const symbolByName = new Map()
21
- for (const s of symbols) {
22
- const name = s.keywords[0]
23
- if (!name || name.length < 3) continue
24
- if (!symbolByName.has(name)) symbolByName.set(name, { syms: [], ...buildMatcher(name) })
25
- symbolByName.get(name).syms.push(s)
30
+ /** 纯 latin 符号名按整词查倒排;其余(CJK / 混合 / 带 `$`)保留正则回退,语义与旧实现一致。 */
31
+ export function buildSymbolIndex(store) {
32
+ const latin = new Map() // lowerName -> symbol[]
33
+ const other = []
34
+ const otherByName = new Map()
35
+ for (const entry of store.allEntries()) {
36
+ if (!entry || entry.type !== 'symbol') continue
37
+ const name = Array.isArray(entry.keywords) ? entry.keywords[0] : undefined
38
+ if (typeof name !== 'string' || name.length < 3) continue
39
+ const key = name.toLowerCase()
40
+ if (LATIN_NAME.test(name)) {
41
+ const bucket = latin.get(key)
42
+ if (bucket) bucket.push(entry)
43
+ else latin.set(key, [entry])
44
+ continue
45
+ }
46
+ const existing = otherByName.get(key)
47
+ if (existing) {
48
+ existing.syms.push(entry)
49
+ continue
50
+ }
51
+ const { re } = buildMatcher(name)
52
+ const record = { reGlobal: new RegExp(re.source, 'g'), syms: [entry] }
53
+ otherByName.set(key, record)
54
+ other.push(record)
26
55
  }
56
+ return { latin, other }
57
+ }
27
58
 
28
- let links = 0
29
- for (const doc of docs) {
30
- const linked = new Set()
31
- // 结构词项也参与链接:terms 覆盖整个 chunk,符号在后半段被提及时同样能链上(与检索同源)
32
- const haystack = `${doc.title || ''} ${doc.summary || ''} ${doc.terms || ''} ${doc.keywords ? doc.keywords.join(' ') : ''}`.toLowerCase()
33
- for (const [, entry] of symbolByName) {
34
- const hit = entry.re ? entry.re.test(haystack) : haystack.includes(entry.lower)
35
- for (const s of entry.syms) {
36
- if (!hit) break
37
- const before = linked.size
38
- linked.add(s.id)
39
- if (linked.size > before) links++
40
- }
41
- }
42
- const before = Array.isArray(doc.linkedSymbols) ? doc.linkedSymbols.join('\u0000') : ''
43
- const next = [...linked]
44
- if (before === next.join('\u0000')) continue
45
- doc.linkedSymbols = next.length ? next : undefined
46
- // 链接是在**符号**落盘那一刻算出来的,此时 doc 的 shard 往往不是脏的;不标脏就只存在于内存,
47
- // 下次进程启动重新加载后链接全部丢失("文档先索引、符号后到"的正常顺序)。
48
- if (typeof store.markFile === 'function' && typeof store.fileRecord === 'function' && doc.sourcePath) {
49
- const record = store.fileRecord(doc.sourcePath)
50
- if (record) store.markFile(doc.sourcePath, record)
59
+ /** 每 store 一份符号索引,按 `store.entriesVersion` 失效(任何 entry 变更即重建)。 */
60
+ const indexCache = new WeakMap()
61
+
62
+ function symbolIndexOf(store) {
63
+ const version = store.entriesVersion
64
+ const cached = indexCache.get(store)
65
+ if (cached && cached.version === version) return cached.index
66
+ const index = buildSymbolIndex(store)
67
+ indexCache.set(store, { version, index })
68
+ return index
69
+ }
70
+
71
+ /**
72
+ * 解算一条 doc entry 提到的符号,按相关度取前 `limit` 个。
73
+ *
74
+ * 判据与旧 linkEntries 同源(title / summary / terms / keywords 四个字段),
75
+ * 排序为:命中次数 → 名字长度 → id(稳定序)。旧实现交给消费者的前 5 个是符号表的
76
+ * 插入序,即「任意 5 个」,这也是本次一并修掉的行为。
77
+ *
78
+ * @param {object} store 带 `allEntries()` 与 `entriesVersion` 的 ProjectMemoryStore
79
+ * @param {object} doc 待解算的 doc entry
80
+ * @param {number} limit 最多返回多少个符号 entry
81
+ * @returns {object[]} 符号 entry 列表(可能为空)
82
+ */
83
+ export function resolveLinkedSymbols(store, doc, limit = 5) {
84
+ if (!store || !doc || doc.type !== 'doc' || !(limit > 0)) return []
85
+ const haystack = `${doc.title || ''} ${doc.summary || ''} ${doc.terms || ''} ${
86
+ Array.isArray(doc.keywords) ? doc.keywords.join(' ') : ''
87
+ }`.toLowerCase()
88
+ if (!haystack.trim()) return []
89
+
90
+ const { latin, other } = symbolIndexOf(store)
91
+ const hits = new Map() // symbol entry -> 命中次数
92
+
93
+ if (latin.size) {
94
+ const counts = new Map() // token -> 在 haystack 中的出现次数
95
+ for (const token of haystack.match(LATIN_WORD) || []) counts.set(token, (counts.get(token) || 0) + 1)
96
+ for (const [token, n] of counts) {
97
+ const bucket = latin.get(token)
98
+ if (!bucket) continue
99
+ for (const symbol of bucket) hits.set(symbol, (hits.get(symbol) || 0) + n)
51
100
  }
52
101
  }
53
- return links
102
+ for (const { reGlobal, syms } of other) {
103
+ reGlobal.lastIndex = 0
104
+ const n = haystack.match(reGlobal)?.length || 0
105
+ if (!n) continue
106
+ for (const symbol of syms) hits.set(symbol, (hits.get(symbol) || 0) + n)
107
+ }
108
+ if (!hits.size) return []
109
+
110
+ const nameOf = (e) => (Array.isArray(e.keywords) && e.keywords[0]) || e.title || ''
111
+ return [...hits.entries()]
112
+ .sort((a, b) => {
113
+ if (b[1] !== a[1]) return b[1] - a[1]
114
+ const delta = nameOf(b[0]).length - nameOf(a[0]).length
115
+ if (delta !== 0) return delta
116
+ return String(a[0].id).localeCompare(String(b[0].id))
117
+ })
118
+ .slice(0, limit)
119
+ .map(([entry]) => entry)
54
120
  }
package/src/store.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { createHash, randomUUID } from 'node:crypto'
2
- import { mkdirSync, readFileSync, readdirSync, renameSync, statSync, unlinkSync, writeFileSync } from 'node:fs'
2
+ import { mkdirSync, readFileSync, readdirSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync } from 'node:fs'
3
3
  import path from 'node:path'
4
- import { rankEntries, rankExperience, tokenize, tokenizeRaw, extractCjkPhrases, makeSearchText } from './util/search.js'
4
+ import { rankExperience, tokenize, tokenizeRaw, extractCjkPhrases, makeSearchText } from './util/search.js'
5
5
  import { backfillDerivedTriggers } from './readiness.js'
6
6
 
7
7
  const FORMAT_FILE = 'format.json'
@@ -13,9 +13,39 @@ const BINDING_FILE = 'binding.json'
13
13
  const WATCH_FILE = 'watch.json'
14
14
  const INSIGHTS_FILE = 'insights.json'
15
15
  const SHARDS_DIR = 'shards'
16
+ const GITIGNORE_FILE = '.gitignore'
17
+ /** ≤0.5.10 的 TS 增强结果缓存目录;0.5.11 起废弃并自愈清理。 */
18
+ const LEGACY_TYPE_CACHE_DIR = 'type-cache'
16
19
 
17
20
  const storeCache = new Map()
21
+ /** 条目数上限(兜底)。 */
18
22
  const STORE_CACHE_MAX = 32
23
+ /**
24
+ * 驻留内存的估算上限。只数"几个 store"是不够的:单个大仓库 store 加载后堆占用实测
25
+ * 130–200MB,32 个就是几 GB。按 store 的条目数与读入文本量估算,约 2.5KB/entry。
26
+ */
27
+ const STORE_CACHE_MAX_BYTES = 256 * 1024 * 1024
28
+ /** 每轮 save 最多补写多少个待压实的老分片(有界,避免升级后一次重写整库)。 */
29
+ const COMPACT_BATCH = 200
30
+
31
+ /** 单个 store 的驻留估算(驻留文本量 vs 条目数换算,取大者)。 */
32
+ function estimateResidentBytes(store) {
33
+ let entries = 0
34
+ for (const list of Object.values(store.entries)) entries += list.length
35
+ return Math.max(store._residentChars || 0, entries * 2500)
36
+ }
37
+
38
+ /** 按 LRU 逐出,直到同时满足条数与字节预算;刚加入的 keepKey 即使超预算也保留。 */
39
+ function evictStoreCache(keepKey) {
40
+ let total = 0
41
+ for (const store of storeCache.values()) total += estimateResidentBytes(store)
42
+ while (storeCache.size > STORE_CACHE_MAX || total > STORE_CACHE_MAX_BYTES) {
43
+ const oldest = storeCache.keys().next().value
44
+ if (oldest === undefined || oldest === keepKey) break
45
+ total -= estimateResidentBytes(storeCache.get(oldest))
46
+ storeCache.delete(oldest)
47
+ }
48
+ }
19
49
 
20
50
  /** 已就"无法迁移的旧 store"告警过的目录:避免每次 load() 都刷一行。 */
21
51
  const migrationWarned = new Set()
@@ -25,13 +55,37 @@ function isRecord(value) {
25
55
  return Boolean(value) && typeof value === 'object' && !Array.isArray(value)
26
56
  }
27
57
 
28
- function loadJson(filePath, fallback) {
58
+ /**
59
+ * 落盘的 entry 只保留**不可推导的事实**。两个派生字段永远不写盘:
60
+ * - `linkedSymbols`:跨实体派生(取决于符号表当前状态),读取期由 src/link.js 解算;
61
+ * - `searchText`:自派生(只依赖本 entry),由 `allEntries()` 在内存里物化。
62
+ * 旧 store 里的这两个字段在加载时剥掉,于是下一次写盘自然压实。
63
+ * 实测两者在一个真实大仓库 store 里合计约 245MB(链接 222.8MB + searchText 22.7MB)。
64
+ */
65
+ const PERSISTED_DERIVED = ['linkedSymbols', 'searchText', 'typeSig']
66
+
67
+ /** 原地剥离派生字段(加载路径用,省掉一次分配)。 */
68
+ function stripPersistedDerived(entry) {
69
+ if (!entry || typeof entry !== 'object') return entry
70
+ for (const key of PERSISTED_DERIVED) if (key in entry) delete entry[key]
71
+ return entry
72
+ }
73
+
74
+ /** 返回不含派生字段的副本(写盘路径用,不能动内存里的对象)。 */
75
+ function withoutPersistedDerived(entry) {
76
+ const out = { ...entry }
77
+ for (const key of PERSISTED_DERIVED) delete out[key]
78
+ return out
79
+ }
80
+
81
+ function loadJson(filePath, fallback, sizeSink) {
29
82
  let raw
30
83
  try {
31
84
  raw = readFileSync(filePath, 'utf8')
32
85
  } catch {
33
86
  return fallback
34
87
  }
88
+ if (sizeSink) sizeSink.bytes += raw.length
35
89
  try {
36
90
  return JSON.parse(raw)
37
91
  } catch {
@@ -83,8 +137,13 @@ export class ProjectMemoryStore {
83
137
  this._dirtyBinding = false
84
138
  this._dirtyWatch = false
85
139
  this._formatWritten = false
86
- this._version = 0
87
140
  this._idfCache = null
141
+ /** entries 的变更计数:IDF 缓存与符号索引(读取期链接)都据此失效。 */
142
+ this._entriesVersion = 0
143
+ /** 待压实的老分片(≤0.5.10 落盘时带 linkedSymbols/searchText);save() 每轮有界补写。 */
144
+ this._compactQueue = new Set()
145
+ /** 从磁盘读入的 JSON 文本量(UTF-16 字符数),作为驻留内存的估算基数。 */
146
+ this._residentChars = 0
88
147
  }
89
148
 
90
149
  load() {
@@ -92,17 +151,21 @@ export class ProjectMemoryStore {
92
151
  const hot = storeCache.get(key)
93
152
  // 缓存命中即返回:`hot === this` 时再读一遍盘会静默丢弃本实例尚未 save() 的变更
94
153
  // (_loadSharded/_loadInsights 会重新赋值 experience/tasks/insights…)。
95
- if (hot) return hot
154
+ if (hot) {
155
+ // LRU:命中挪到队尾,避免"热的先被逐出、冷的常驻"。
156
+ storeCache.delete(key)
157
+ storeCache.set(key, hot)
158
+ return hot
159
+ }
96
160
  this._migrateLegacyIfNeeded()
97
161
  this._loadSharded()
98
162
  this._loadInsights()
163
+ this._removeLegacyTypeCache()
164
+ // 有些路径只读不写(审计 jsonl 直接写在 store 目录里),所以这里也补一次自我忽略,
165
+ // 让老版本建出来的 store 在第一次 load 就补上。
166
+ this.ensureSelfIgnore()
99
167
  storeCache.set(key, this)
100
- // 只保留最近打开的项目:长期跨多项目运行时不至于无限增长(被逐出只是下次重新读盘)
101
- while (storeCache.size > STORE_CACHE_MAX) {
102
- const oldest = storeCache.keys().next().value
103
- if (oldest === key) break
104
- storeCache.delete(oldest)
105
- }
168
+ evictStoreCache(key)
106
169
  return this
107
170
  }
108
171
 
@@ -160,7 +223,7 @@ export class ProjectMemoryStore {
160
223
  }
161
224
  mkdirSync(path.join(this.dir, SHARDS_DIR), { recursive: true })
162
225
  for (const rel of Object.keys(files)) {
163
- writeJsonAtomic(shardRelPath(this.dir, rel), { relPath: rel, record: files[rel], entries: entries[rel] || [] })
226
+ writeJsonAtomic(shardRelPath(this.dir, rel), { relPath: rel, record: files[rel], entries: (entries[rel] || []).map(withoutPersistedDerived) })
164
227
  }
165
228
  writeJsonAtomic(formatPath, { version: 2, layout: 'sharded' })
166
229
  for (const stale of [legacyEntriesPath, path.join(this.dir, INDEX_FILE)]) {
@@ -180,21 +243,45 @@ export class ProjectMemoryStore {
180
243
  } catch {
181
244
  shardNames = []
182
245
  }
246
+ const sizeSink = { bytes: 0 }
183
247
  for (const name of shardNames) {
184
- const shard = loadJson(path.join(this.dir, SHARDS_DIR, name), null)
248
+ const shard = loadJson(path.join(this.dir, SHARDS_DIR, name), null, sizeSink)
185
249
  if (!shard || typeof shard.relPath !== 'string' || !isRecord(shard.record)) continue
186
250
  this.files[shard.relPath] = shard.record
187
251
  // 畸形 shard(entries 被写成对象/null)不能让 allEntries() 在 `for…of` 上抛错,
188
252
  // 否则一个坏文件会拖垮整个进程的每一次读取。
189
- this.entries[shard.relPath] = Array.isArray(shard.entries) ? shard.entries.filter(isRecord) : []
190
- }
191
- this.experience = loadJson(path.join(this.dir, EXPERIENCE_FILE), [])
192
- this.tasks = loadJson(path.join(this.dir, TASKS_FILE), [])
193
- this.binding = loadJson(path.join(this.dir, BINDING_FILE), {})
194
- this.watchlist = loadJson(path.join(this.dir, WATCH_FILE), [])
253
+ const list = Array.isArray(shard.entries) ? shard.entries.filter(isRecord) : []
254
+ // 老分片带着派生字段:内存里立刻剥掉,并排队等 save() 把磁盘上也压实。
255
+ if (list.some((e) => PERSISTED_DERIVED.some((k) => k in e))) this._compactQueue.add(shard.relPath)
256
+ this.entries[shard.relPath] = list.map(stripPersistedDerived)
257
+ }
258
+ this._entriesVersion++
259
+ this.experience = loadJson(path.join(this.dir, EXPERIENCE_FILE), [], sizeSink)
260
+ this.tasks = loadJson(path.join(this.dir, TASKS_FILE), [], sizeSink)
261
+ this.binding = loadJson(path.join(this.dir, BINDING_FILE), {}, sizeSink)
262
+ this.watchlist = loadJson(path.join(this.dir, WATCH_FILE), [], sizeSink)
263
+ this._residentChars = sizeSink.bytes
195
264
  this._formatWritten = existsSafe(path.join(this.dir, FORMAT_FILE))
196
265
  }
197
266
 
267
+ /**
268
+ * 清掉 ≤0.5.10 留下的 `type-cache/` 目录。
269
+ *
270
+ * 它按内容哈希缓存 TS 增强结果,但三个增强入口(lazy 的 `fs/observed`、watch 轮询、
271
+ * `index_repo`)**都只在"文件已变更并重新索引"之后**才触发,此时内容哈希必然是新值——
272
+ * 这个缓存永远命中不了。实测本仓库残留 9827 个文件(`du` 41MB,内容其实 9.2MB,约 31MB
273
+ * 是 4KB 块开销)。这里做一次自愈清理;删的是纯缓存,不丢任何事实。
274
+ */
275
+ _removeLegacyTypeCache() {
276
+ const dir = path.join(this.dir, LEGACY_TYPE_CACHE_DIR)
277
+ if (!existsSafe(dir)) return
278
+ try {
279
+ rmSync(dir, { recursive: true, force: true })
280
+ } catch {
281
+ // 权限/占用导致删不掉也不影响 store 本身
282
+ }
283
+ }
284
+
198
285
  // ---- v0.5 insights:project 级 insight 文档(任务级在 task.insights[]) ----
199
286
  // 迁移语义(旧版 store 的兼容策略):
200
287
  // v0.4 experience.json 仍由 remember/forget/query_memory 服务,不删除;
@@ -286,11 +373,35 @@ export class ProjectMemoryStore {
286
373
  }
287
374
  }
288
375
 
376
+ /**
377
+ * store 目录**自我忽略**:在目录里放一个内容为 `*` 的 `.gitignore`。
378
+ *
379
+ * 为什么不写进用户的 `.gitignore`:那是用户的文件,插件不该改;而且"忘了加"的代价
380
+ * 是一次误提交。为什么这样就够:git 会读取工作区里**任意**目录下的 `.gitignore`,
381
+ * 而 `*` 连这个 `.gitignore` 自己一起命中,于是 `git status` / `git add -A` 里整棵树
382
+ * 都不出现,用户一个字都不用写(`git check-ignore -v` 可复核)。
383
+ *
384
+ * 顺带的好处:被忽略的文件不会被 `git clean -fd` 删除(未跟踪且未忽略的会被删)。
385
+ * 想把记忆跟着仓库提交:`git add -f .dsh-project-memory`——已跟踪的文件不受忽略规则影响。
386
+ */
387
+ ensureSelfIgnore() {
388
+ if (!existsSafe(this.dir)) return
389
+ const file = path.join(this.dir, GITIGNORE_FILE)
390
+ if (existsSafe(file)) return
391
+ try {
392
+ writeFileSync(file, '# dsh-project-memory: local by default. `git add -f` to commit it.\n*\n')
393
+ } catch {
394
+ // 旁路:写不进去不影响存储本身
395
+ }
396
+ }
397
+
289
398
  save() {
290
399
  // 没有脏数据就不落盘。watch 每轮对每个根都无条件 commit → save;照旧执行的话,
291
- // 末尾的 `_version++` + `_idfCache = null` 会打在跨实例共享的 store 上,
292
- // 等于每 15 秒清空一次 IDF 缓存,废掉查询侧的 IDF 复用(v0.3.4 的 20x)。
400
+ // 末尾的 ID 缓存失效会打在跨实例共享的 store 上,等于每轮清空一次 IDF 复用(v0.3.4 的 20x)。
401
+ // 待压实队列不算"脏数据",但它需要有界推进,所以也走这条落盘路径。
402
+ const compacting = this._compactQueue.size > 0
293
403
  const dirty =
404
+ compacting ||
294
405
  this._dirtyShards.size > 0 ||
295
406
  this._removedShards.size > 0 ||
296
407
  this._dirtyExperience ||
@@ -302,15 +413,27 @@ export class ProjectMemoryStore {
302
413
  if (dirty) mkdirSync(this.dir, { recursive: true })
303
414
  // 崩溃遗留的 *.tmp 无论有没有脏数据都顺手清掉(两次 readdir,自带 try/catch)
304
415
  this.cleanStaleTmp()
416
+ // 自我忽略也在无脏数据时执行:老版本建出来的 store 会在下一次 save 时补上。
417
+ this.ensureSelfIgnore()
305
418
  if (!dirty) return
306
419
  if (!this._formatWritten) {
307
420
  writeJsonAtomic(path.join(this.dir, FORMAT_FILE), { version: 2, layout: 'sharded' })
308
421
  this._formatWritten = true
309
422
  }
423
+ // 存量压实:≤0.5.10 的分片带着派生字段,而那些字段只在分片被重写时才会从磁盘消失。
424
+ // 每轮最多补写 COMPACT_BATCH 个,让升级后的 store 在后续任意一次 save(watch 轮询、
425
+ // 索引、写入)里自动收敛,而不是永远停在旧体积。
426
+ let compactBudget = COMPACT_BATCH
427
+ for (const rel of this._compactQueue) {
428
+ if (compactBudget <= 0) break
429
+ compactBudget--
430
+ this._compactQueue.delete(rel)
431
+ if (this.files[rel]) this._dirtyShards.add(rel)
432
+ }
310
433
  for (const rel of this._dirtyShards) {
311
434
  if (this.files[rel]) {
312
435
  mkdirSync(path.join(this.dir, SHARDS_DIR), { recursive: true })
313
- writeJsonAtomic(shardRelPath(this.dir, rel), { relPath: rel, record: this.files[rel], entries: this.entries[rel] || [] })
436
+ writeJsonAtomic(shardRelPath(this.dir, rel), { relPath: rel, record: this.files[rel], entries: (this.entries[rel] || []).map(withoutPersistedDerived) })
314
437
  } else {
315
438
  this._removedShards.add(rel)
316
439
  }
@@ -346,16 +469,16 @@ export class ProjectMemoryStore {
346
469
  writeJsonAtomic(path.join(this.dir, WATCH_FILE), this.watchlist)
347
470
  this._dirtyWatch = false
348
471
  }
349
- this._version++
350
- this._idfCache = null
351
472
  }
352
473
 
353
474
  getIdfCache() {
354
- if (this._idfCache && this._idfCache.version === this._version) {
475
+ // IDF 只依赖 entries(title/keywords/summary),所以按 entries 的变更计数失效:
476
+ // 只写经验/insight 或只做存量压实的 save 不会再无谓地重建 IDF。
477
+ if (this._idfCache && this._idfCache.version === this._entriesVersion) {
355
478
  return this._idfCache.idf
356
479
  }
357
480
  const idf = this._rebuildIdf()
358
- this._idfCache = { version: this._version, idf }
481
+ this._idfCache = { version: this._entriesVersion, idf }
359
482
  return idf
360
483
  }
361
484
 
@@ -410,11 +533,17 @@ export class ProjectMemoryStore {
410
533
 
411
534
  setEntries(relPath, entries) {
412
535
  if (entries.length) {
413
- const enriched = entries.map((e) => ({ ...e, searchText: makeSearchText(e) }))
536
+ // searchText 在内存里物化(写入/检索都靠它),但 save() 不会把它写盘。
537
+ const enriched = entries.map((e) => {
538
+ const out = stripPersistedDerived({ ...e })
539
+ out.searchText = makeSearchText(out)
540
+ return out
541
+ })
414
542
  this.entries[relPath] = enriched
415
543
  } else {
416
544
  delete this.entries[relPath]
417
545
  }
546
+ this._entriesVersion++
418
547
  this._dirtyShards.add(relPath)
419
548
  this._removedShards.delete(relPath)
420
549
  }
@@ -423,23 +552,34 @@ export class ProjectMemoryStore {
423
552
  if (relPath in this.files) {
424
553
  delete this.files[relPath]
425
554
  delete this.entries[relPath]
555
+ this._entriesVersion++
426
556
  this._dirtyShards.add(relPath)
427
557
  this._removedShards.add(relPath)
428
558
  }
429
559
  }
430
560
 
561
+ /** entries 的变更计数:派生缓存(符号索引)据此失效。 */
562
+ get entriesVersion() {
563
+ return this._entriesVersion
564
+ }
565
+
566
+ /** 还有多少个老分片等着被 save() 压实(0 = 存量已收敛)。 */
567
+ get pendingCompaction() {
568
+ return this._compactQueue.size
569
+ }
570
+
431
571
  allEntries() {
432
572
  const out = []
433
573
  for (const list of Object.values(this.entries)) {
434
- for (const entry of list) out.push(entry)
574
+ for (const entry of list) {
575
+ // searchText 不落盘,首次用到时按需物化并挂在对象上(同一 entry 只算一次)。
576
+ if (entry.searchText === undefined) entry.searchText = makeSearchText(entry)
577
+ out.push(entry)
578
+ }
435
579
  }
436
580
  return out
437
581
  }
438
582
 
439
- searchEntries(query, limit = 8) {
440
- return rankEntries(this.allEntries(), query, limit)
441
- }
442
-
443
583
  addExperience({ problem, solution, sourceFile }) {
444
584
  const existing = this.findSupersede(problem)
445
585
  const now = new Date().toISOString()
package/src/symbols.js CHANGED
@@ -1,4 +1,5 @@
1
1
  import { readFileSync } from 'node:fs'
2
+ import { oneLineDeclaration } from './util/text.js'
2
3
 
3
4
  const JS_LIKE = new Set(['.js', '.mjs', '.cjs', '.ts', '.mts', '.cts', '.jsx', '.tsx'])
4
5
  const PYTHON = new Set(['.py'])
@@ -465,7 +466,8 @@ function buildSymbol(matched, relPath, rawLine, lineNo) {
465
466
  type: 'symbol',
466
467
  title: `${matched.name} (${matched.kind})`,
467
468
  keywords: [matched.name, matched.kind],
468
- text: identity,
469
+ // interface/type 的 typeSig 可能是整个类型体;这里限长成一行,避免符号条目被撑大。
470
+ text: oneLineDeclaration(identity),
469
471
  }
470
472
  }
471
473
 
@@ -3,7 +3,6 @@ import path from 'node:path'
3
3
  import { assertIndexRoot, assertReadableFile, assertSafeRoot, findProjectRoot, memoryRootFor, sessionMemoryRootOrNull, sha256OfFile, storeKey } from '../util/fs.js'
4
4
  import { buildDocEntries } from '../doc-pipeline.js'
5
5
  import { docEntriesNeedBackfill } from '../doc-index.js'
6
- import { linkEntries } from '../link.js'
7
6
  import { ProjectMemoryStore } from '../store.js'
8
7
 
9
8
  export function indexDocTool(ctx, config) {
@@ -76,7 +75,6 @@ export function indexDocTool(ctx, config) {
76
75
  return store.commit((s) => {
77
76
  s.setEntries(rel, entries)
78
77
  s.markFile(rel, { sha256: hash, size, type: 'doc', indexedAt: new Date().toISOString() })
79
- linkEntries(s)
80
78
  const preview = entries
81
79
  .map((e) => ` - ${e.title} @ ${rel}:${e.sourceLine}`)
82
80
  .join('\n')
@@ -33,8 +33,8 @@ export async function indexRepository(ctx, config, root, { reindex = false, allo
33
33
 
34
34
  const flushBatch = () => {
35
35
  if (!batch.length) return
36
- // 中间批次不做 unseen 清理、不重建链接(最后一批统一做)——否则每批都要遍历整个 store。
37
- commitFileUpdates(store, { updates: batch, link: false })
36
+ // 中间批次不做 unseen 清理(最后一批统一做)——否则每批都要遍历整个 store。
37
+ commitFileUpdates(store, { updates: batch })
38
38
  batch = []
39
39
  }
40
40
 
@@ -74,12 +74,11 @@ export async function indexRepository(ctx, config, root, { reindex = false, allo
74
74
  }
75
75
  }
76
76
 
77
- // 收尾:写完最后一批 + 清理本轮未见到的旧条目 + 重建链接。
77
+ // 收尾:写完最后一批 + 清理本轮未见到的旧条目。
78
78
  // 截断时**不做 unseen 清理**:没扫到的文件不等于被删了。
79
79
  const { removed } = commitFileUpdates(store, {
80
80
  updates: batch,
81
81
  unseen: truncated ? null : seen,
82
- link: true,
83
82
  })
84
83
  const stats = store.stats()
85
84
  let report =
@@ -6,6 +6,7 @@ import { expandQuery } from '../llm.js'
6
6
  import { resolveRoute } from '../llm-route.js'
7
7
  import { GlobalStore, cfgInsight, defaultGlobalFile, recordHit } from '../insight-store.js'
8
8
  import { recallItems } from '../recall.js'
9
+ import { resolveLinkedSymbols } from '../link.js'
9
10
  import { truncate } from '../util/text.js'
10
11
  import { stepContent, stepStatus } from '../util/task-view.js'
11
12
  function toAbs(root, rel) {
@@ -59,10 +60,6 @@ export function queryMemoryTool(ctx, config) {
59
60
  const queries = config.llmQueryExpansion
60
61
  ? await expandQuery(ctx.llm, args.query, config.expansionCount, { route: resolveRoute(exec, config) })
61
62
  : [args.query]
62
- const symbolById = new Map()
63
- for (const e of store.allEntries()) {
64
- if (e.type === 'symbol') symbolById.set(e.id, e)
65
- }
66
63
 
67
64
  // 会话绑定决定 task 级 insight 的可见性;global 级始终可见。
68
65
  const sessionId = exec?.agent?.session?.id
@@ -121,12 +118,12 @@ export function queryMemoryTool(ctx, config) {
121
118
  // 条目状态占位(可证伪状态机落地前恒为 exact)。
122
119
  // 先立字段,后续状态机到位时只改值、不改输出契约。
123
120
  lines.push(`### ${e.title} (score: ${rel})\n- source: ${absSource}\n- status: ${e.status || 'exact'}\n${summaryLine}`)
124
- if (e.type === 'doc' && Array.isArray(e.linkedSymbols) && e.linkedSymbols.length) {
125
- const refs = e.linkedSymbols.slice(0, 5).map((id) => {
126
- const s = symbolById.get(id)
127
- return s ? `${s.title} @ ${toAbs(root, s.sourcePath)}:${s.sourceLine}` : id
128
- })
129
- lines.push(`- references: ${refs.join('; ')}`)
121
+ if (e.type === 'doc') {
122
+ // 读取期解算:链接不落盘,按当前符号表排序取前 5(见 src/link.js)
123
+ const refs = resolveLinkedSymbols(store, e, 5).map(
124
+ (s) => `${s.title} @ ${toAbs(root, s.sourcePath)}:${s.sourceLine}`,
125
+ )
126
+ if (refs.length) lines.push(`- references: ${refs.join('; ')}`)
130
127
  }
131
128
  }
132
129
  }