@yolk_vat-y/dsh-project-memory 0.5.9 → 0.5.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +117 -0
- package/README.md +16 -16
- package/README.zh-CN.md +16 -16
- package/cordis.patch.yml +1 -1
- package/package.json +2 -2
- package/scripts/bench-synthetic.mjs +2 -4
- package/src/enhancer.js +13 -42
- package/src/index-pipeline.js +5 -4
- package/src/index.js +1 -1
- package/src/link.js +104 -38
- package/src/store.js +172 -32
- package/src/symbols.js +3 -1
- package/src/tools/index-doc.js +0 -2
- package/src/tools/index-repo.js +3 -4
- package/src/tools/query-memory.js +7 -10
- package/src/util/fs.js +59 -20
- package/src/util/search.js +4 -9
- package/src/util/text.js +16 -0
- package/src/watch.js +40 -12
package/src/link.js
CHANGED
|
@@ -1,54 +1,120 @@
|
|
|
1
|
+
// doc → symbol 交叉引用:**读取期解算**,不再物化进 entry。
|
|
2
|
+
//
|
|
3
|
+
// 旧实现(≤0.5.10)在每次索引提交时把「本 chunk 命中的所有符号 id」写进
|
|
4
|
+
// `entry.linkedSymbols` 并落盘。实测一个真实 TypeScript 仓库(本仓库的 361MB store):
|
|
5
|
+
// 17174 个 chunk 共 3,948,420 个链接槽位,只对应 15,456 个符号,单 chunk 最多 2231 个;
|
|
6
|
+
// 而唯一的消费者 `query_memory` 只读前 5 个——存了消费量的 46 倍。
|
|
7
|
+
//
|
|
8
|
+
// 更根本的问题是它把**跨实体派生关系**固化进了源记录:doc 链接的有效性取决于符号表的
|
|
9
|
+
// 当前状态,而符号是独立写入的。于是必须追失效(`markFile` 标脏、"文档先索引、符号后到
|
|
10
|
+
// 则链接丢失"),并且每个索引提交点都要对整库做 O(chunks × symbols) 全表重扫。
|
|
11
|
+
//
|
|
12
|
+
// 现在:entry 只存事实;链接在查询时用每 store 的符号索引解算,成本 O(本 chunk 词数),
|
|
13
|
+
// 且天然反映当前符号表——后索引的符号也能链上,`markFile` 那套失效机制随之删除。
|
|
14
|
+
|
|
1
15
|
const LATIN_NAME = /^[A-Za-z0-9_]+$/
|
|
2
16
|
const CJK_CHAR = /[\u3400-\u9fff\uf900-\ufaff\u3040-\u309f\u30a0-\u30ff\uac00-\ud7af]/
|
|
17
|
+
/** 与 latin 边界后视 `[a-z0-9_$]` 同字符集:切出的词直接可查倒排表。 */
|
|
18
|
+
const LATIN_WORD = /[a-z0-9_$]+/g
|
|
3
19
|
|
|
4
20
|
function buildMatcher(name) {
|
|
5
21
|
const lower = name.toLowerCase()
|
|
6
22
|
if (LATIN_NAME.test(name)) {
|
|
7
|
-
return {
|
|
23
|
+
return { re: new RegExp(`(?<![a-z0-9_$])${lower}(?![a-z0-9_$])`) }
|
|
8
24
|
}
|
|
9
25
|
const escaped = lower.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
|
|
10
26
|
// 尾部统一挡 CJK + 字母数字下划线,防止纯 CJK 名误链混合后缀、混合名误链更长后缀
|
|
11
|
-
return {
|
|
27
|
+
return { re: new RegExp(`${escaped}(?!${CJK_CHAR.source})(?![a-z0-9_$])`) }
|
|
12
28
|
}
|
|
13
29
|
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
const
|
|
17
|
-
const
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
if (
|
|
25
|
-
|
|
30
|
+
/** 纯 latin 符号名按整词查倒排;其余(CJK / 混合 / 带 `$`)保留正则回退,语义与旧实现一致。 */
|
|
31
|
+
export function buildSymbolIndex(store) {
|
|
32
|
+
const latin = new Map() // lowerName -> symbol[]
|
|
33
|
+
const other = []
|
|
34
|
+
const otherByName = new Map()
|
|
35
|
+
for (const entry of store.allEntries()) {
|
|
36
|
+
if (!entry || entry.type !== 'symbol') continue
|
|
37
|
+
const name = Array.isArray(entry.keywords) ? entry.keywords[0] : undefined
|
|
38
|
+
if (typeof name !== 'string' || name.length < 3) continue
|
|
39
|
+
const key = name.toLowerCase()
|
|
40
|
+
if (LATIN_NAME.test(name)) {
|
|
41
|
+
const bucket = latin.get(key)
|
|
42
|
+
if (bucket) bucket.push(entry)
|
|
43
|
+
else latin.set(key, [entry])
|
|
44
|
+
continue
|
|
45
|
+
}
|
|
46
|
+
const existing = otherByName.get(key)
|
|
47
|
+
if (existing) {
|
|
48
|
+
existing.syms.push(entry)
|
|
49
|
+
continue
|
|
50
|
+
}
|
|
51
|
+
const { re } = buildMatcher(name)
|
|
52
|
+
const record = { reGlobal: new RegExp(re.source, 'g'), syms: [entry] }
|
|
53
|
+
otherByName.set(key, record)
|
|
54
|
+
other.push(record)
|
|
26
55
|
}
|
|
56
|
+
return { latin, other }
|
|
57
|
+
}
|
|
27
58
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
59
|
+
/** 每 store 一份符号索引,按 `store.entriesVersion` 失效(任何 entry 变更即重建)。 */
|
|
60
|
+
const indexCache = new WeakMap()
|
|
61
|
+
|
|
62
|
+
function symbolIndexOf(store) {
|
|
63
|
+
const version = store.entriesVersion
|
|
64
|
+
const cached = indexCache.get(store)
|
|
65
|
+
if (cached && cached.version === version) return cached.index
|
|
66
|
+
const index = buildSymbolIndex(store)
|
|
67
|
+
indexCache.set(store, { version, index })
|
|
68
|
+
return index
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* 解算一条 doc entry 提到的符号,按相关度取前 `limit` 个。
|
|
73
|
+
*
|
|
74
|
+
* 判据与旧 linkEntries 同源(title / summary / terms / keywords 四个字段),
|
|
75
|
+
* 排序为:命中次数 → 名字长度 → id(稳定序)。旧实现交给消费者的前 5 个是符号表的
|
|
76
|
+
* 插入序,即「任意 5 个」,这也是本次一并修掉的行为。
|
|
77
|
+
*
|
|
78
|
+
* @param {object} store 带 `allEntries()` 与 `entriesVersion` 的 ProjectMemoryStore
|
|
79
|
+
* @param {object} doc 待解算的 doc entry
|
|
80
|
+
* @param {number} limit 最多返回多少个符号 entry
|
|
81
|
+
* @returns {object[]} 符号 entry 列表(可能为空)
|
|
82
|
+
*/
|
|
83
|
+
export function resolveLinkedSymbols(store, doc, limit = 5) {
|
|
84
|
+
if (!store || !doc || doc.type !== 'doc' || !(limit > 0)) return []
|
|
85
|
+
const haystack = `${doc.title || ''} ${doc.summary || ''} ${doc.terms || ''} ${
|
|
86
|
+
Array.isArray(doc.keywords) ? doc.keywords.join(' ') : ''
|
|
87
|
+
}`.toLowerCase()
|
|
88
|
+
if (!haystack.trim()) return []
|
|
89
|
+
|
|
90
|
+
const { latin, other } = symbolIndexOf(store)
|
|
91
|
+
const hits = new Map() // symbol entry -> 命中次数
|
|
92
|
+
|
|
93
|
+
if (latin.size) {
|
|
94
|
+
const counts = new Map() // token -> 在 haystack 中的出现次数
|
|
95
|
+
for (const token of haystack.match(LATIN_WORD) || []) counts.set(token, (counts.get(token) || 0) + 1)
|
|
96
|
+
for (const [token, n] of counts) {
|
|
97
|
+
const bucket = latin.get(token)
|
|
98
|
+
if (!bucket) continue
|
|
99
|
+
for (const symbol of bucket) hits.set(symbol, (hits.get(symbol) || 0) + n)
|
|
51
100
|
}
|
|
52
101
|
}
|
|
53
|
-
|
|
102
|
+
for (const { reGlobal, syms } of other) {
|
|
103
|
+
reGlobal.lastIndex = 0
|
|
104
|
+
const n = haystack.match(reGlobal)?.length || 0
|
|
105
|
+
if (!n) continue
|
|
106
|
+
for (const symbol of syms) hits.set(symbol, (hits.get(symbol) || 0) + n)
|
|
107
|
+
}
|
|
108
|
+
if (!hits.size) return []
|
|
109
|
+
|
|
110
|
+
const nameOf = (e) => (Array.isArray(e.keywords) && e.keywords[0]) || e.title || ''
|
|
111
|
+
return [...hits.entries()]
|
|
112
|
+
.sort((a, b) => {
|
|
113
|
+
if (b[1] !== a[1]) return b[1] - a[1]
|
|
114
|
+
const delta = nameOf(b[0]).length - nameOf(a[0]).length
|
|
115
|
+
if (delta !== 0) return delta
|
|
116
|
+
return String(a[0].id).localeCompare(String(b[0].id))
|
|
117
|
+
})
|
|
118
|
+
.slice(0, limit)
|
|
119
|
+
.map(([entry]) => entry)
|
|
54
120
|
}
|
package/src/store.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { createHash, randomUUID } from 'node:crypto'
|
|
2
|
-
import { mkdirSync, readFileSync, readdirSync, renameSync, statSync, unlinkSync, writeFileSync } from 'node:fs'
|
|
2
|
+
import { mkdirSync, readFileSync, readdirSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync } from 'node:fs'
|
|
3
3
|
import path from 'node:path'
|
|
4
|
-
import {
|
|
4
|
+
import { rankExperience, tokenize, tokenizeRaw, extractCjkPhrases, makeSearchText } from './util/search.js'
|
|
5
5
|
import { backfillDerivedTriggers } from './readiness.js'
|
|
6
6
|
|
|
7
7
|
const FORMAT_FILE = 'format.json'
|
|
@@ -13,9 +13,39 @@ const BINDING_FILE = 'binding.json'
|
|
|
13
13
|
const WATCH_FILE = 'watch.json'
|
|
14
14
|
const INSIGHTS_FILE = 'insights.json'
|
|
15
15
|
const SHARDS_DIR = 'shards'
|
|
16
|
+
const GITIGNORE_FILE = '.gitignore'
|
|
17
|
+
/** ≤0.5.10 的 TS 增强结果缓存目录;0.5.11 起废弃并自愈清理。 */
|
|
18
|
+
const LEGACY_TYPE_CACHE_DIR = 'type-cache'
|
|
16
19
|
|
|
17
20
|
const storeCache = new Map()
|
|
21
|
+
/** 条目数上限(兜底)。 */
|
|
18
22
|
const STORE_CACHE_MAX = 32
|
|
23
|
+
/**
|
|
24
|
+
* 驻留内存的估算上限。只数"几个 store"是不够的:单个大仓库 store 加载后堆占用实测
|
|
25
|
+
* 130–200MB,32 个就是几 GB。按 store 的条目数与读入文本量估算,约 2.5KB/entry。
|
|
26
|
+
*/
|
|
27
|
+
const STORE_CACHE_MAX_BYTES = 256 * 1024 * 1024
|
|
28
|
+
/** 每轮 save 最多补写多少个待压实的老分片(有界,避免升级后一次重写整库)。 */
|
|
29
|
+
const COMPACT_BATCH = 200
|
|
30
|
+
|
|
31
|
+
/** 单个 store 的驻留估算(驻留文本量 vs 条目数换算,取大者)。 */
|
|
32
|
+
function estimateResidentBytes(store) {
|
|
33
|
+
let entries = 0
|
|
34
|
+
for (const list of Object.values(store.entries)) entries += list.length
|
|
35
|
+
return Math.max(store._residentChars || 0, entries * 2500)
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** 按 LRU 逐出,直到同时满足条数与字节预算;刚加入的 keepKey 即使超预算也保留。 */
|
|
39
|
+
function evictStoreCache(keepKey) {
|
|
40
|
+
let total = 0
|
|
41
|
+
for (const store of storeCache.values()) total += estimateResidentBytes(store)
|
|
42
|
+
while (storeCache.size > STORE_CACHE_MAX || total > STORE_CACHE_MAX_BYTES) {
|
|
43
|
+
const oldest = storeCache.keys().next().value
|
|
44
|
+
if (oldest === undefined || oldest === keepKey) break
|
|
45
|
+
total -= estimateResidentBytes(storeCache.get(oldest))
|
|
46
|
+
storeCache.delete(oldest)
|
|
47
|
+
}
|
|
48
|
+
}
|
|
19
49
|
|
|
20
50
|
/** 已就"无法迁移的旧 store"告警过的目录:避免每次 load() 都刷一行。 */
|
|
21
51
|
const migrationWarned = new Set()
|
|
@@ -25,13 +55,37 @@ function isRecord(value) {
|
|
|
25
55
|
return Boolean(value) && typeof value === 'object' && !Array.isArray(value)
|
|
26
56
|
}
|
|
27
57
|
|
|
28
|
-
|
|
58
|
+
/**
|
|
59
|
+
* 落盘的 entry 只保留**不可推导的事实**。两个派生字段永远不写盘:
|
|
60
|
+
* - `linkedSymbols`:跨实体派生(取决于符号表当前状态),读取期由 src/link.js 解算;
|
|
61
|
+
* - `searchText`:自派生(只依赖本 entry),由 `allEntries()` 在内存里物化。
|
|
62
|
+
* 旧 store 里的这两个字段在加载时剥掉,于是下一次写盘自然压实。
|
|
63
|
+
* 实测两者在一个真实大仓库 store 里合计约 245MB(链接 222.8MB + searchText 22.7MB)。
|
|
64
|
+
*/
|
|
65
|
+
const PERSISTED_DERIVED = ['linkedSymbols', 'searchText', 'typeSig']
|
|
66
|
+
|
|
67
|
+
/** 原地剥离派生字段(加载路径用,省掉一次分配)。 */
|
|
68
|
+
function stripPersistedDerived(entry) {
|
|
69
|
+
if (!entry || typeof entry !== 'object') return entry
|
|
70
|
+
for (const key of PERSISTED_DERIVED) if (key in entry) delete entry[key]
|
|
71
|
+
return entry
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** 返回不含派生字段的副本(写盘路径用,不能动内存里的对象)。 */
|
|
75
|
+
function withoutPersistedDerived(entry) {
|
|
76
|
+
const out = { ...entry }
|
|
77
|
+
for (const key of PERSISTED_DERIVED) delete out[key]
|
|
78
|
+
return out
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
function loadJson(filePath, fallback, sizeSink) {
|
|
29
82
|
let raw
|
|
30
83
|
try {
|
|
31
84
|
raw = readFileSync(filePath, 'utf8')
|
|
32
85
|
} catch {
|
|
33
86
|
return fallback
|
|
34
87
|
}
|
|
88
|
+
if (sizeSink) sizeSink.bytes += raw.length
|
|
35
89
|
try {
|
|
36
90
|
return JSON.parse(raw)
|
|
37
91
|
} catch {
|
|
@@ -83,8 +137,13 @@ export class ProjectMemoryStore {
|
|
|
83
137
|
this._dirtyBinding = false
|
|
84
138
|
this._dirtyWatch = false
|
|
85
139
|
this._formatWritten = false
|
|
86
|
-
this._version = 0
|
|
87
140
|
this._idfCache = null
|
|
141
|
+
/** entries 的变更计数:IDF 缓存与符号索引(读取期链接)都据此失效。 */
|
|
142
|
+
this._entriesVersion = 0
|
|
143
|
+
/** 待压实的老分片(≤0.5.10 落盘时带 linkedSymbols/searchText);save() 每轮有界补写。 */
|
|
144
|
+
this._compactQueue = new Set()
|
|
145
|
+
/** 从磁盘读入的 JSON 文本量(UTF-16 字符数),作为驻留内存的估算基数。 */
|
|
146
|
+
this._residentChars = 0
|
|
88
147
|
}
|
|
89
148
|
|
|
90
149
|
load() {
|
|
@@ -92,17 +151,21 @@ export class ProjectMemoryStore {
|
|
|
92
151
|
const hot = storeCache.get(key)
|
|
93
152
|
// 缓存命中即返回:`hot === this` 时再读一遍盘会静默丢弃本实例尚未 save() 的变更
|
|
94
153
|
// (_loadSharded/_loadInsights 会重新赋值 experience/tasks/insights…)。
|
|
95
|
-
if (hot)
|
|
154
|
+
if (hot) {
|
|
155
|
+
// LRU:命中挪到队尾,避免"热的先被逐出、冷的常驻"。
|
|
156
|
+
storeCache.delete(key)
|
|
157
|
+
storeCache.set(key, hot)
|
|
158
|
+
return hot
|
|
159
|
+
}
|
|
96
160
|
this._migrateLegacyIfNeeded()
|
|
97
161
|
this._loadSharded()
|
|
98
162
|
this._loadInsights()
|
|
163
|
+
this._removeLegacyTypeCache()
|
|
164
|
+
// 有些路径只读不写(审计 jsonl 直接写在 store 目录里),所以这里也补一次自我忽略,
|
|
165
|
+
// 让老版本建出来的 store 在第一次 load 就补上。
|
|
166
|
+
this.ensureSelfIgnore()
|
|
99
167
|
storeCache.set(key, this)
|
|
100
|
-
|
|
101
|
-
while (storeCache.size > STORE_CACHE_MAX) {
|
|
102
|
-
const oldest = storeCache.keys().next().value
|
|
103
|
-
if (oldest === key) break
|
|
104
|
-
storeCache.delete(oldest)
|
|
105
|
-
}
|
|
168
|
+
evictStoreCache(key)
|
|
106
169
|
return this
|
|
107
170
|
}
|
|
108
171
|
|
|
@@ -160,7 +223,7 @@ export class ProjectMemoryStore {
|
|
|
160
223
|
}
|
|
161
224
|
mkdirSync(path.join(this.dir, SHARDS_DIR), { recursive: true })
|
|
162
225
|
for (const rel of Object.keys(files)) {
|
|
163
|
-
writeJsonAtomic(shardRelPath(this.dir, rel), { relPath: rel, record: files[rel], entries: entries[rel] || [] })
|
|
226
|
+
writeJsonAtomic(shardRelPath(this.dir, rel), { relPath: rel, record: files[rel], entries: (entries[rel] || []).map(withoutPersistedDerived) })
|
|
164
227
|
}
|
|
165
228
|
writeJsonAtomic(formatPath, { version: 2, layout: 'sharded' })
|
|
166
229
|
for (const stale of [legacyEntriesPath, path.join(this.dir, INDEX_FILE)]) {
|
|
@@ -180,21 +243,45 @@ export class ProjectMemoryStore {
|
|
|
180
243
|
} catch {
|
|
181
244
|
shardNames = []
|
|
182
245
|
}
|
|
246
|
+
const sizeSink = { bytes: 0 }
|
|
183
247
|
for (const name of shardNames) {
|
|
184
|
-
const shard = loadJson(path.join(this.dir, SHARDS_DIR, name), null)
|
|
248
|
+
const shard = loadJson(path.join(this.dir, SHARDS_DIR, name), null, sizeSink)
|
|
185
249
|
if (!shard || typeof shard.relPath !== 'string' || !isRecord(shard.record)) continue
|
|
186
250
|
this.files[shard.relPath] = shard.record
|
|
187
251
|
// 畸形 shard(entries 被写成对象/null)不能让 allEntries() 在 `for…of` 上抛错,
|
|
188
252
|
// 否则一个坏文件会拖垮整个进程的每一次读取。
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
this.
|
|
253
|
+
const list = Array.isArray(shard.entries) ? shard.entries.filter(isRecord) : []
|
|
254
|
+
// 老分片带着派生字段:内存里立刻剥掉,并排队等 save() 把磁盘上也压实。
|
|
255
|
+
if (list.some((e) => PERSISTED_DERIVED.some((k) => k in e))) this._compactQueue.add(shard.relPath)
|
|
256
|
+
this.entries[shard.relPath] = list.map(stripPersistedDerived)
|
|
257
|
+
}
|
|
258
|
+
this._entriesVersion++
|
|
259
|
+
this.experience = loadJson(path.join(this.dir, EXPERIENCE_FILE), [], sizeSink)
|
|
260
|
+
this.tasks = loadJson(path.join(this.dir, TASKS_FILE), [], sizeSink)
|
|
261
|
+
this.binding = loadJson(path.join(this.dir, BINDING_FILE), {}, sizeSink)
|
|
262
|
+
this.watchlist = loadJson(path.join(this.dir, WATCH_FILE), [], sizeSink)
|
|
263
|
+
this._residentChars = sizeSink.bytes
|
|
195
264
|
this._formatWritten = existsSafe(path.join(this.dir, FORMAT_FILE))
|
|
196
265
|
}
|
|
197
266
|
|
|
267
|
+
/**
|
|
268
|
+
* 清掉 ≤0.5.10 留下的 `type-cache/` 目录。
|
|
269
|
+
*
|
|
270
|
+
* 它按内容哈希缓存 TS 增强结果,但三个增强入口(lazy 的 `fs/observed`、watch 轮询、
|
|
271
|
+
* `index_repo`)**都只在"文件已变更并重新索引"之后**才触发,此时内容哈希必然是新值——
|
|
272
|
+
* 这个缓存永远命中不了。实测本仓库残留 9827 个文件(`du` 41MB,内容其实 9.2MB,约 31MB
|
|
273
|
+
* 是 4KB 块开销)。这里做一次自愈清理;删的是纯缓存,不丢任何事实。
|
|
274
|
+
*/
|
|
275
|
+
_removeLegacyTypeCache() {
|
|
276
|
+
const dir = path.join(this.dir, LEGACY_TYPE_CACHE_DIR)
|
|
277
|
+
if (!existsSafe(dir)) return
|
|
278
|
+
try {
|
|
279
|
+
rmSync(dir, { recursive: true, force: true })
|
|
280
|
+
} catch {
|
|
281
|
+
// 权限/占用导致删不掉也不影响 store 本身
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
|
|
198
285
|
// ---- v0.5 insights:project 级 insight 文档(任务级在 task.insights[]) ----
|
|
199
286
|
// 迁移语义(旧版 store 的兼容策略):
|
|
200
287
|
// v0.4 experience.json 仍由 remember/forget/query_memory 服务,不删除;
|
|
@@ -286,11 +373,35 @@ export class ProjectMemoryStore {
|
|
|
286
373
|
}
|
|
287
374
|
}
|
|
288
375
|
|
|
376
|
+
/**
|
|
377
|
+
* store 目录**自我忽略**:在目录里放一个内容为 `*` 的 `.gitignore`。
|
|
378
|
+
*
|
|
379
|
+
* 为什么不写进用户的 `.gitignore`:那是用户的文件,插件不该改;而且"忘了加"的代价
|
|
380
|
+
* 是一次误提交。为什么这样就够:git 会读取工作区里**任意**目录下的 `.gitignore`,
|
|
381
|
+
* 而 `*` 连这个 `.gitignore` 自己一起命中,于是 `git status` / `git add -A` 里整棵树
|
|
382
|
+
* 都不出现,用户一个字都不用写(`git check-ignore -v` 可复核)。
|
|
383
|
+
*
|
|
384
|
+
* 顺带的好处:被忽略的文件不会被 `git clean -fd` 删除(未跟踪且未忽略的会被删)。
|
|
385
|
+
* 想把记忆跟着仓库提交:`git add -f .dsh-project-memory`——已跟踪的文件不受忽略规则影响。
|
|
386
|
+
*/
|
|
387
|
+
ensureSelfIgnore() {
|
|
388
|
+
if (!existsSafe(this.dir)) return
|
|
389
|
+
const file = path.join(this.dir, GITIGNORE_FILE)
|
|
390
|
+
if (existsSafe(file)) return
|
|
391
|
+
try {
|
|
392
|
+
writeFileSync(file, '# dsh-project-memory: local by default. `git add -f` to commit it.\n*\n')
|
|
393
|
+
} catch {
|
|
394
|
+
// 旁路:写不进去不影响存储本身
|
|
395
|
+
}
|
|
396
|
+
}
|
|
397
|
+
|
|
289
398
|
save() {
|
|
290
399
|
// 没有脏数据就不落盘。watch 每轮对每个根都无条件 commit → save;照旧执行的话,
|
|
291
|
-
// 末尾的
|
|
292
|
-
//
|
|
400
|
+
// 末尾的 ID 缓存失效会打在跨实例共享的 store 上,等于每轮清空一次 IDF 复用(v0.3.4 的 20x)。
|
|
401
|
+
// 待压实队列不算"脏数据",但它需要有界推进,所以也走这条落盘路径。
|
|
402
|
+
const compacting = this._compactQueue.size > 0
|
|
293
403
|
const dirty =
|
|
404
|
+
compacting ||
|
|
294
405
|
this._dirtyShards.size > 0 ||
|
|
295
406
|
this._removedShards.size > 0 ||
|
|
296
407
|
this._dirtyExperience ||
|
|
@@ -302,15 +413,27 @@ export class ProjectMemoryStore {
|
|
|
302
413
|
if (dirty) mkdirSync(this.dir, { recursive: true })
|
|
303
414
|
// 崩溃遗留的 *.tmp 无论有没有脏数据都顺手清掉(两次 readdir,自带 try/catch)
|
|
304
415
|
this.cleanStaleTmp()
|
|
416
|
+
// 自我忽略也在无脏数据时执行:老版本建出来的 store 会在下一次 save 时补上。
|
|
417
|
+
this.ensureSelfIgnore()
|
|
305
418
|
if (!dirty) return
|
|
306
419
|
if (!this._formatWritten) {
|
|
307
420
|
writeJsonAtomic(path.join(this.dir, FORMAT_FILE), { version: 2, layout: 'sharded' })
|
|
308
421
|
this._formatWritten = true
|
|
309
422
|
}
|
|
423
|
+
// 存量压实:≤0.5.10 的分片带着派生字段,而那些字段只在分片被重写时才会从磁盘消失。
|
|
424
|
+
// 每轮最多补写 COMPACT_BATCH 个,让升级后的 store 在后续任意一次 save(watch 轮询、
|
|
425
|
+
// 索引、写入)里自动收敛,而不是永远停在旧体积。
|
|
426
|
+
let compactBudget = COMPACT_BATCH
|
|
427
|
+
for (const rel of this._compactQueue) {
|
|
428
|
+
if (compactBudget <= 0) break
|
|
429
|
+
compactBudget--
|
|
430
|
+
this._compactQueue.delete(rel)
|
|
431
|
+
if (this.files[rel]) this._dirtyShards.add(rel)
|
|
432
|
+
}
|
|
310
433
|
for (const rel of this._dirtyShards) {
|
|
311
434
|
if (this.files[rel]) {
|
|
312
435
|
mkdirSync(path.join(this.dir, SHARDS_DIR), { recursive: true })
|
|
313
|
-
writeJsonAtomic(shardRelPath(this.dir, rel), { relPath: rel, record: this.files[rel], entries: this.entries[rel] || [] })
|
|
436
|
+
writeJsonAtomic(shardRelPath(this.dir, rel), { relPath: rel, record: this.files[rel], entries: (this.entries[rel] || []).map(withoutPersistedDerived) })
|
|
314
437
|
} else {
|
|
315
438
|
this._removedShards.add(rel)
|
|
316
439
|
}
|
|
@@ -346,16 +469,16 @@ export class ProjectMemoryStore {
|
|
|
346
469
|
writeJsonAtomic(path.join(this.dir, WATCH_FILE), this.watchlist)
|
|
347
470
|
this._dirtyWatch = false
|
|
348
471
|
}
|
|
349
|
-
this._version++
|
|
350
|
-
this._idfCache = null
|
|
351
472
|
}
|
|
352
473
|
|
|
353
474
|
getIdfCache() {
|
|
354
|
-
|
|
475
|
+
// IDF 只依赖 entries(title/keywords/summary),所以按 entries 的变更计数失效:
|
|
476
|
+
// 只写经验/insight 或只做存量压实的 save 不会再无谓地重建 IDF。
|
|
477
|
+
if (this._idfCache && this._idfCache.version === this._entriesVersion) {
|
|
355
478
|
return this._idfCache.idf
|
|
356
479
|
}
|
|
357
480
|
const idf = this._rebuildIdf()
|
|
358
|
-
this._idfCache = { version: this.
|
|
481
|
+
this._idfCache = { version: this._entriesVersion, idf }
|
|
359
482
|
return idf
|
|
360
483
|
}
|
|
361
484
|
|
|
@@ -410,11 +533,17 @@ export class ProjectMemoryStore {
|
|
|
410
533
|
|
|
411
534
|
setEntries(relPath, entries) {
|
|
412
535
|
if (entries.length) {
|
|
413
|
-
|
|
536
|
+
// searchText 在内存里物化(写入/检索都靠它),但 save() 不会把它写盘。
|
|
537
|
+
const enriched = entries.map((e) => {
|
|
538
|
+
const out = stripPersistedDerived({ ...e })
|
|
539
|
+
out.searchText = makeSearchText(out)
|
|
540
|
+
return out
|
|
541
|
+
})
|
|
414
542
|
this.entries[relPath] = enriched
|
|
415
543
|
} else {
|
|
416
544
|
delete this.entries[relPath]
|
|
417
545
|
}
|
|
546
|
+
this._entriesVersion++
|
|
418
547
|
this._dirtyShards.add(relPath)
|
|
419
548
|
this._removedShards.delete(relPath)
|
|
420
549
|
}
|
|
@@ -423,23 +552,34 @@ export class ProjectMemoryStore {
|
|
|
423
552
|
if (relPath in this.files) {
|
|
424
553
|
delete this.files[relPath]
|
|
425
554
|
delete this.entries[relPath]
|
|
555
|
+
this._entriesVersion++
|
|
426
556
|
this._dirtyShards.add(relPath)
|
|
427
557
|
this._removedShards.add(relPath)
|
|
428
558
|
}
|
|
429
559
|
}
|
|
430
560
|
|
|
561
|
+
/** entries 的变更计数:派生缓存(符号索引)据此失效。 */
|
|
562
|
+
get entriesVersion() {
|
|
563
|
+
return this._entriesVersion
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
/** 还有多少个老分片等着被 save() 压实(0 = 存量已收敛)。 */
|
|
567
|
+
get pendingCompaction() {
|
|
568
|
+
return this._compactQueue.size
|
|
569
|
+
}
|
|
570
|
+
|
|
431
571
|
allEntries() {
|
|
432
572
|
const out = []
|
|
433
573
|
for (const list of Object.values(this.entries)) {
|
|
434
|
-
for (const entry of list)
|
|
574
|
+
for (const entry of list) {
|
|
575
|
+
// searchText 不落盘,首次用到时按需物化并挂在对象上(同一 entry 只算一次)。
|
|
576
|
+
if (entry.searchText === undefined) entry.searchText = makeSearchText(entry)
|
|
577
|
+
out.push(entry)
|
|
578
|
+
}
|
|
435
579
|
}
|
|
436
580
|
return out
|
|
437
581
|
}
|
|
438
582
|
|
|
439
|
-
searchEntries(query, limit = 8) {
|
|
440
|
-
return rankEntries(this.allEntries(), query, limit)
|
|
441
|
-
}
|
|
442
|
-
|
|
443
583
|
addExperience({ problem, solution, sourceFile }) {
|
|
444
584
|
const existing = this.findSupersede(problem)
|
|
445
585
|
const now = new Date().toISOString()
|
package/src/symbols.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { readFileSync } from 'node:fs'
|
|
2
|
+
import { oneLineDeclaration } from './util/text.js'
|
|
2
3
|
|
|
3
4
|
const JS_LIKE = new Set(['.js', '.mjs', '.cjs', '.ts', '.mts', '.cts', '.jsx', '.tsx'])
|
|
4
5
|
const PYTHON = new Set(['.py'])
|
|
@@ -465,7 +466,8 @@ function buildSymbol(matched, relPath, rawLine, lineNo) {
|
|
|
465
466
|
type: 'symbol',
|
|
466
467
|
title: `${matched.name} (${matched.kind})`,
|
|
467
468
|
keywords: [matched.name, matched.kind],
|
|
468
|
-
|
|
469
|
+
// interface/type 的 typeSig 可能是整个类型体;这里限长成一行,避免符号条目被撑大。
|
|
470
|
+
text: oneLineDeclaration(identity),
|
|
469
471
|
}
|
|
470
472
|
}
|
|
471
473
|
|
package/src/tools/index-doc.js
CHANGED
|
@@ -3,7 +3,6 @@ import path from 'node:path'
|
|
|
3
3
|
import { assertIndexRoot, assertReadableFile, assertSafeRoot, findProjectRoot, memoryRootFor, sessionMemoryRootOrNull, sha256OfFile, storeKey } from '../util/fs.js'
|
|
4
4
|
import { buildDocEntries } from '../doc-pipeline.js'
|
|
5
5
|
import { docEntriesNeedBackfill } from '../doc-index.js'
|
|
6
|
-
import { linkEntries } from '../link.js'
|
|
7
6
|
import { ProjectMemoryStore } from '../store.js'
|
|
8
7
|
|
|
9
8
|
export function indexDocTool(ctx, config) {
|
|
@@ -76,7 +75,6 @@ export function indexDocTool(ctx, config) {
|
|
|
76
75
|
return store.commit((s) => {
|
|
77
76
|
s.setEntries(rel, entries)
|
|
78
77
|
s.markFile(rel, { sha256: hash, size, type: 'doc', indexedAt: new Date().toISOString() })
|
|
79
|
-
linkEntries(s)
|
|
80
78
|
const preview = entries
|
|
81
79
|
.map((e) => ` - ${e.title} @ ${rel}:${e.sourceLine}`)
|
|
82
80
|
.join('\n')
|
package/src/tools/index-repo.js
CHANGED
|
@@ -33,8 +33,8 @@ export async function indexRepository(ctx, config, root, { reindex = false, allo
|
|
|
33
33
|
|
|
34
34
|
const flushBatch = () => {
|
|
35
35
|
if (!batch.length) return
|
|
36
|
-
// 中间批次不做 unseen
|
|
37
|
-
commitFileUpdates(store, { updates: batch
|
|
36
|
+
// 中间批次不做 unseen 清理(最后一批统一做)——否则每批都要遍历整个 store。
|
|
37
|
+
commitFileUpdates(store, { updates: batch })
|
|
38
38
|
batch = []
|
|
39
39
|
}
|
|
40
40
|
|
|
@@ -74,12 +74,11 @@ export async function indexRepository(ctx, config, root, { reindex = false, allo
|
|
|
74
74
|
}
|
|
75
75
|
}
|
|
76
76
|
|
|
77
|
-
// 收尾:写完最后一批 +
|
|
77
|
+
// 收尾:写完最后一批 + 清理本轮未见到的旧条目。
|
|
78
78
|
// 截断时**不做 unseen 清理**:没扫到的文件不等于被删了。
|
|
79
79
|
const { removed } = commitFileUpdates(store, {
|
|
80
80
|
updates: batch,
|
|
81
81
|
unseen: truncated ? null : seen,
|
|
82
|
-
link: true,
|
|
83
82
|
})
|
|
84
83
|
const stats = store.stats()
|
|
85
84
|
let report =
|
|
@@ -6,6 +6,7 @@ import { expandQuery } from '../llm.js'
|
|
|
6
6
|
import { resolveRoute } from '../llm-route.js'
|
|
7
7
|
import { GlobalStore, cfgInsight, defaultGlobalFile, recordHit } from '../insight-store.js'
|
|
8
8
|
import { recallItems } from '../recall.js'
|
|
9
|
+
import { resolveLinkedSymbols } from '../link.js'
|
|
9
10
|
import { truncate } from '../util/text.js'
|
|
10
11
|
import { stepContent, stepStatus } from '../util/task-view.js'
|
|
11
12
|
function toAbs(root, rel) {
|
|
@@ -59,10 +60,6 @@ export function queryMemoryTool(ctx, config) {
|
|
|
59
60
|
const queries = config.llmQueryExpansion
|
|
60
61
|
? await expandQuery(ctx.llm, args.query, config.expansionCount, { route: resolveRoute(exec, config) })
|
|
61
62
|
: [args.query]
|
|
62
|
-
const symbolById = new Map()
|
|
63
|
-
for (const e of store.allEntries()) {
|
|
64
|
-
if (e.type === 'symbol') symbolById.set(e.id, e)
|
|
65
|
-
}
|
|
66
63
|
|
|
67
64
|
// 会话绑定决定 task 级 insight 的可见性;global 级始终可见。
|
|
68
65
|
const sessionId = exec?.agent?.session?.id
|
|
@@ -121,12 +118,12 @@ export function queryMemoryTool(ctx, config) {
|
|
|
121
118
|
// 条目状态占位(可证伪状态机落地前恒为 exact)。
|
|
122
119
|
// 先立字段,后续状态机到位时只改值、不改输出契约。
|
|
123
120
|
lines.push(`### ${e.title} (score: ${rel})\n- source: ${absSource}\n- status: ${e.status || 'exact'}\n${summaryLine}`)
|
|
124
|
-
if (e.type === 'doc'
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
lines.push(`- references: ${refs.join('; ')}`)
|
|
121
|
+
if (e.type === 'doc') {
|
|
122
|
+
// 读取期解算:链接不落盘,按当前符号表排序取前 5(见 src/link.js)
|
|
123
|
+
const refs = resolveLinkedSymbols(store, e, 5).map(
|
|
124
|
+
(s) => `${s.title} @ ${toAbs(root, s.sourcePath)}:${s.sourceLine}`,
|
|
125
|
+
)
|
|
126
|
+
if (refs.length) lines.push(`- references: ${refs.join('; ')}`)
|
|
130
127
|
}
|
|
131
128
|
}
|
|
132
129
|
}
|