@yolk_vat-y/dsh-project-memory 0.5.3 → 0.5.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,396 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * 独立基准:给一个真实项目路径,量「冷索引 / 冷加载 / 热查询 / 单文件热重索引 / 存储体积」。
4
+ *
5
+ * 特点:
6
+ * - **不需要运行 dsh**,只 import 插件自己的 src 模块,走的就是线上那条索引与检索路径;
7
+ * - **不碰被测项目**:结果写进临时目录,跑完删除(`--keep` 可保留);
8
+ * - 索引期零模型调用、零网络请求,所以在任何机器上都能复现。
9
+ *
10
+ * 用法:
11
+ * node scripts/bench.mjs [projectPath] [--json] [--samples 100] [--no-pdf] [--keep] [--max-files 20000]
12
+ * node scripts/bench.mjs ./my-project --queries my-queries.json # 可选:带标注集算 hit@k / MRR
13
+ *
14
+ * `--queries` 的 JSON 形如:
15
+ * [{ "query": "jwt token validation", "expect": "src/utils/auth.ts" }, ...]
16
+ * `expect` 按子串匹配结果的 sourcePath / title / id(大小写不敏感)。
17
+ */
18
+ import { cpSync, existsSync, mkdtempSync, readdirSync, rmSync, statSync } from 'node:fs'
19
+ import os from 'node:os'
20
+ import path from 'node:path'
21
+ import { performance } from 'node:perf_hooks'
22
+
23
+ import { ProjectMemoryStore } from '../src/store.js'
24
+ import { buildDocEntries } from '../src/doc-pipeline.js'
25
+ import { scanSymbols } from '../src/symbols.js'
26
+ import { rankEntriesStreaming } from '../src/util/search.js'
27
+ import {
28
+ isSupportedCode,
29
+ isSupportedDoc,
30
+ readFileForIndex,
31
+ relativePath,
32
+ walkDir,
33
+ } from '../src/util/fs.js'
34
+
35
+ // ---------- CLI ----------
36
+ const argv = process.argv.slice(2)
37
+ const opts = { json: false, samples: 100, pdf: true, keep: false, maxFiles: 20000, queries: null, target: null }
38
+ for (let i = 0; i < argv.length; i++) {
39
+ const a = argv[i]
40
+ if (a === '--json') opts.json = true
41
+ else if (a === '--no-pdf') opts.pdf = false
42
+ else if (a === '--keep') opts.keep = true
43
+ else if (a === '--help' || a === '-h') opts.help = true
44
+ else if (a === '--samples') opts.samples = Math.max(1, Number(argv[++i]) || 100)
45
+ else if (a === '--max-files') opts.maxFiles = Math.max(1, Number(argv[++i]) || 20000)
46
+ else if (a === '--queries') opts.queries = argv[++i]
47
+ else if (a.startsWith('--')) fail(`unknown option: ${a}`)
48
+ else if (!opts.target) opts.target = a
49
+ else fail(`unexpected argument: ${a}`)
50
+ }
51
+
52
+ function fail(msg) {
53
+ console.error(`bench: ${msg}`)
54
+ console.error('usage: node scripts/bench.mjs [projectPath] [--json] [--samples N] [--no-pdf] [--keep] [--max-files N] [--queries file.json]')
55
+ process.exit(2)
56
+ }
57
+
58
+ if (opts.help) {
59
+ console.log('usage: node scripts/bench.mjs [projectPath] [--json] [--samples N] [--no-pdf] [--keep] [--max-files N] [--queries file.json]')
60
+ process.exit(0)
61
+ }
62
+
63
+ const root = path.resolve(opts.target || process.cwd())
64
+ if (!existsSync(root) || !statSync(root).isDirectory()) fail(`not a directory: ${root}`)
65
+
66
+ // ---------- helpers ----------
67
+ const now = () => performance.now()
68
+ const round = (n, d = 2) => Number(n.toFixed(d))
69
+
70
+ function timeSync(fn) {
71
+ const t0 = now()
72
+ const value = fn()
73
+ return { ms: now() - t0, value }
74
+ }
75
+
76
+ async function timeAsync(fn) {
77
+ const t0 = now()
78
+ const value = await fn()
79
+ return { ms: now() - t0, value }
80
+ }
81
+
82
+ function stats(xs) {
83
+ if (!xs.length) return { n: 0, p50: 0, p95: 0, max: 0, avg: 0 }
84
+ const s = [...xs].sort((a, b) => a - b)
85
+ const at = (q) => s[Math.min(s.length - 1, Math.floor(q * s.length))]
86
+ return {
87
+ n: s.length,
88
+ p50: round(at(0.5), 3),
89
+ p95: round(at(0.95), 3),
90
+ max: round(s[s.length - 1], 3),
91
+ avg: round(s.reduce((a, b) => a + b, 0) / s.length, 3),
92
+ }
93
+ }
94
+
95
+ function dirSize(dir) {
96
+ let bytes = 0
97
+ let files = 0
98
+ const stack = [dir]
99
+ while (stack.length) {
100
+ let entries
101
+ try {
102
+ entries = readdirSync(stack.pop(), { withFileTypes: true })
103
+ } catch {
104
+ continue
105
+ }
106
+ for (const e of entries) {
107
+ const p = path.join(e.parentPath || e.path, e.name)
108
+ if (e.isDirectory()) stack.push(p)
109
+ else {
110
+ try {
111
+ bytes += statSync(p).size
112
+ files++
113
+ } catch {
114
+ /* ignore */
115
+ }
116
+ }
117
+ }
118
+ }
119
+ return { bytes, files }
120
+ }
121
+
122
+ const kb = (n) => `${(n / 1024).toFixed(1)} KB`
123
+ const mb = (n) => `${(n / 1024 / 1024).toFixed(2)} MB`
124
+
125
+ // ---------- 1. 扫描 ----------
126
+ const all = walkDir(root)
127
+ const targets = []
128
+ let skippedPdf = 0
129
+ for (const abs of all) {
130
+ const ext = path.extname(abs).toLowerCase()
131
+ const code = isSupportedCode(ext)
132
+ const doc = isSupportedDoc(ext)
133
+ if (!code && !doc) continue
134
+ if (ext === '.pdf' && !opts.pdf) {
135
+ skippedPdf++
136
+ continue
137
+ }
138
+ targets.push({ abs, rel: relativePath(root, abs), code, doc, size: statSync(abs).size })
139
+ }
140
+ const truncated = targets.length > opts.maxFiles
141
+ const files = truncated ? targets.slice(0, opts.maxFiles) : targets
142
+
143
+ // ---------- 2. 冷索引(读+哈希 / 抽取 / 落盘 三段分开计时)----------
144
+ const tmpStore = mkdtempSync(path.join(os.tmpdir(), 'pm-bench-'))
145
+ const store = new ProjectMemoryStore(tmpStore).load()
146
+
147
+ const readTimes = []
148
+ const extractTimes = []
149
+ const perFileTimes = []
150
+ const updates = []
151
+ let dumps = 0
152
+ let errors = 0
153
+ const errorsList = []
154
+
155
+ const indexStart = now()
156
+ for (const f of files) {
157
+ let read
158
+ try {
159
+ read = timeSync(() => readFileForIndex(f.abs))
160
+ } catch (err) {
161
+ errors++
162
+ if (errorsList.length < 5) errorsList.push(`${f.rel}: ${err.message}`)
163
+ continue
164
+ }
165
+ readTimes.push(read.ms)
166
+
167
+ const ex = await timeAsync(async () => {
168
+ if (f.code) return scanSymbols(f.rel, f.abs, read.value.buffer.toString('utf8'))
169
+ return buildDocEntries(f.rel, f.abs, {
170
+ chunkChars: 3000,
171
+ maxChunks: 40,
172
+ maxFileSizeMb: 50,
173
+ maxPdfPages: 1000,
174
+ })
175
+ })
176
+ extractTimes.push(ex.ms)
177
+ perFileTimes.push(read.ms + ex.ms)
178
+ if (ex.value === null) {
179
+ dumps++
180
+ continue
181
+ }
182
+ updates.push({
183
+ rel: f.rel,
184
+ record: { sha256: read.value.hash, size: read.value.size, type: f.code ? 'code' : 'doc', indexedAt: new Date().toISOString() },
185
+ entries: ex.value,
186
+ extractMs: ex.ms,
187
+ })
188
+ }
189
+
190
+ const commit = timeSync(() => {
191
+ store.commit((s) => {
192
+ for (const u of updates) {
193
+ s.markFile(u.rel, u.record)
194
+ s.setEntries(u.rel, u.entries)
195
+ }
196
+ })
197
+ })
198
+ const indexMs = now() - indexStart
199
+
200
+ const entries = store.allEntries()
201
+ const fileCount = Object.keys(store.files).length
202
+
203
+ // ---------- 3. 存储体积 + 冷加载 ----------
204
+ const onDisk = dirSize(tmpStore)
205
+ // 真冷加载:store 有进程内缓存(同目录第二次 load() 直接命中),复制到新目录才是一次真正的冷启动
206
+ const coldDir = `${tmpStore}-cold`
207
+ cpSync(tmpStore, coldDir, { recursive: true })
208
+ const coldLoad = timeSync(() => new ProjectMemoryStore(coldDir).load())
209
+
210
+ // ---------- 4. 热查询 / 冷查询 ----------
211
+ const queryPool = entries.filter((e) => typeof e.title === 'string' && e.title.trim().length >= 4)
212
+ const step = Math.max(1, Math.floor(queryPool.length / opts.samples))
213
+ const queries = []
214
+ for (let i = 0; i < queryPool.length && queries.length < opts.samples; i += step) queries.push(queryPool[i].title.trim())
215
+
216
+ const idfCold = timeSync(() => store.getIdfCache())
217
+ const idf = idfCold.value
218
+ const hotTimes = []
219
+ let firstQueryHits = []
220
+ for (const q of queries) {
221
+ const t = timeSync(() => rankEntriesStreaming(entries, [q], idf, 8))
222
+ hotTimes.push(t.ms)
223
+ if (!firstQueryHits.length) firstQueryHits = t.value
224
+ }
225
+ const coldQuery = timeSync(() => {
226
+ store._idfCache = null // 强制走「写入后首次查询」的重建路径
227
+ const rebuilt = store.getIdfCache()
228
+ return rankEntriesStreaming(entries, [queries[0] || 'memory'], rebuilt, 8)
229
+ })
230
+
231
+ // ---------- 5. 单文件热重索引(watch 那条路径)----------
232
+ const codeRels = Object.keys(store.files).filter((rel) => store.files[rel].type === 'code')
233
+ .sort((a, b) => store.files[a].size - store.files[b].size)
234
+ // 按体积均匀取样,别只取最小的那几个(否则「热重索引」会好看得离谱)
235
+ const want = Math.min(20, codeRels.length)
236
+ const sampleRels = []
237
+ for (let i = 0; i < want; i++) sampleRels.push(codeRels[Math.floor((i + 0.5) * codeRels.length / want)])
238
+ const reindexTimes = []
239
+ for (const rel of sampleRels) {
240
+ const abs = path.join(root, rel)
241
+ try {
242
+ const t = await timeAsync(async () => {
243
+ const r = readFileForIndex(abs)
244
+ const ents = scanSymbols(rel, abs, r.buffer.toString('utf8'))
245
+ store.commit((s) => {
246
+ s.markFile(rel, { sha256: r.hash, size: r.size, type: 'code', indexedAt: new Date().toISOString() })
247
+ s.setEntries(rel, ents)
248
+ })
249
+ })
250
+ reindexTimes.push(t.ms)
251
+ } catch {
252
+ /* 跳过读不到的文件 */
253
+ }
254
+ }
255
+
256
+ // ---------- 6. 整 chunk 词项 vs ≤300 字摘要的覆盖 ----------
257
+ const docEntries = entries.filter((e) => e.type === 'doc' && typeof e.terms === 'string' && e.terms)
258
+ let coverageSum = 0
259
+ for (const e of docEntries) {
260
+ const terms = new Set(e.terms.split(/\s+/).filter(Boolean))
261
+ if (!terms.size) continue
262
+ const summary = String(e.summary || '').toLowerCase()
263
+ let hit = 0
264
+ for (const t of terms) if (summary.includes(t)) hit++
265
+ coverageSum += hit / terms.size
266
+ }
267
+ const coverage = docEntries.length ? coverageSum / docEntries.length : null
268
+
269
+ // ---------- 7. 可选:带标注查询集 ----------
270
+ let quality = null
271
+ if (opts.queries) {
272
+ const qs = JSON.parse(await readFileSafe(path.resolve(opts.queries)))
273
+ if (!Array.isArray(qs) || !qs.length) fail('--queries file must be a non-empty JSON array')
274
+ let hit5 = 0
275
+ let hit10 = 0
276
+ let mrr = 0
277
+ const detail = []
278
+ for (const item of qs) {
279
+ const expect = String(item.expect || '').toLowerCase()
280
+ const results = rankEntriesStreaming(entries, [String(item.query)], idf, 10)
281
+ const rank = results.findIndex((r) => {
282
+ const hay = `${r.entry.sourcePath || ''} ${r.entry.title || ''} ${r.entry.id || ''}`.toLowerCase()
283
+ return expect && hay.includes(expect)
284
+ })
285
+ if (rank === 0 || (rank >= 0 && rank < 5)) hit5++
286
+ if (rank >= 0 && rank < 10) hit10++
287
+ if (rank >= 0) mrr += 1 / (rank + 1)
288
+ detail.push({ query: item.query, expect: item.expect, rank: rank < 0 ? null : rank + 1 })
289
+ }
290
+ quality = {
291
+ queries: qs.length,
292
+ hit5: round((hit5 / qs.length) * 100, 1),
293
+ hit10: round((hit10 / qs.length) * 100, 1),
294
+ mrr: round(mrr / qs.length, 3),
295
+ detail,
296
+ }
297
+ }
298
+
299
+ // ---------- 8. 输出 ----------
300
+ const result = {
301
+ env: {
302
+ node: process.version,
303
+ platform: `${process.platform}-${process.arch}`,
304
+ cpus: os.cpus().length,
305
+ ranAt: new Date().toISOString(),
306
+ },
307
+ project: {
308
+ path: root,
309
+ filesScanned: all.length,
310
+ supported: targets.length,
311
+ indexed: fileCount,
312
+ code: updates.filter((u) => u.record.type === 'code').length,
313
+ doc: updates.filter((u) => u.record.type === 'doc').length,
314
+ dumps: dumps,
315
+ pdfSkipped: skippedPdf,
316
+ truncated,
317
+ errors,
318
+ errorsList,
319
+ },
320
+ index: {
321
+ totalMs: round(indexMs),
322
+ readHashMs: round(readTimes.reduce((a, b) => a + b, 0)),
323
+ extractMs: round(extractTimes.reduce((a, b) => a + b, 0)),
324
+ commitMs: round(commit.ms),
325
+ perFileMs: stats(perFileTimes),
326
+ entries: entries.length,
327
+ },
328
+ store: {
329
+ dir: tmpStore,
330
+ bytes: onDisk.bytes,
331
+ files: onDisk.files,
332
+ bytesPerEntry: entries.length ? Math.round(onDisk.bytes / entries.length) : 0,
333
+ },
334
+ coldLoadMs: round(coldLoad.ms),
335
+ query: {
336
+ samples: queries.length,
337
+ idfRebuildMs: round(idfCold.ms),
338
+ coldMs: round(coldQuery.ms),
339
+ hot: stats(hotTimes),
340
+ },
341
+ hotReindex: stats(reindexTimes),
342
+ termsCoverage: {
343
+ docEntries: docEntries.length,
344
+ meanSummaryCoverage: coverage === null ? null : round(coverage * 100, 1),
345
+ },
346
+ quality,
347
+ }
348
+
349
+ if (opts.json) {
350
+ console.log(JSON.stringify(result, null, 2))
351
+ } else {
352
+ const p = result.project
353
+ console.log(`\n=== dsh-project-memory bench ===`)
354
+ console.log(`project ${root}`)
355
+ console.log(`env ${result.env.node} ${result.env.platform} · ${result.env.cpus} CPU`)
356
+ console.log(`scanned ${p.filesScanned} files → ${p.supported} supported (${p.code} code / ${p.doc} doc)`)
357
+ if (p.dumps) console.log(`skipped ${p.dumps} dump-like files`)
358
+ if (p.pdfSkipped) console.log(`skipped ${p.pdfSkipped} PDFs (--no-pdf)`)
359
+ if (p.truncated) console.log(`truncated to --max-files ${opts.maxFiles}`)
360
+ if (p.errors) console.log(`errors ${p.errors} ${p.errorsList.join(' | ')}`)
361
+ console.log(`\n-- cold index --`)
362
+ console.log(`total ${result.index.totalMs} ms → ${result.index.entries} entries in ${p.indexed} files`)
363
+ console.log(` read+hash ${result.index.readHashMs} ms · extract ${result.index.extractMs} ms · commit ${result.index.commitMs} ms`)
364
+ console.log(` per file p50 ${result.index.perFileMs.p50} ms · p95 ${result.index.perFileMs.p95} ms`)
365
+ console.log(` note: read+hash depends on the OS page cache — run it twice and say which run you quote`)
366
+ console.log(`store ${mb(result.store.bytes)} · ${result.store.bytesPerEntry} bytes/entry · cold load ${result.coldLoadMs} ms`)
367
+ console.log(`\n-- query (shipped scorer: IDF cache + streaming BM25 + CJK phrase boost) --`)
368
+ console.log(`idf rebuild (first query after write) ${result.query.idfRebuildMs} ms`)
369
+ console.log(`cold query (rebuild + rank) ${result.query.coldMs} ms`)
370
+ console.log(`hot query n=${result.query.hot.n} p50 ${result.query.hot.p50} ms · p95 ${result.query.hot.p95} ms · max ${result.query.hot.max} ms`)
371
+ console.log(`\n-- hot single-file re-index (watch path) --`)
372
+ console.log(`n=${result.hotReindex.n} p50 ${result.hotReindex.p50} ms · p95 ${result.hotReindex.p95} ms · max ${result.hotReindex.max} ms`)
373
+ if (result.termsCoverage.docEntries) {
374
+ console.log(`\n-- whole-chunk terms (corpus-dependent; NOT a retrieval-quality metric) --`)
375
+ console.log(`${result.termsCoverage.docEntries} doc chunks · a ≤300-char summary alone already contains ${result.termsCoverage.meanSummaryCoverage}% of each chunk's own literal terms`)
376
+ console.log(`(short chunks inflate this; terms exist for the tail that the summary prefix cannot hold)`)
377
+ }
378
+ if (quality) {
379
+ console.log(`\n-- labeled queries (${quality.queries}) --`)
380
+ console.log(`hit@5 ${quality.hit5}% · hit@10 ${quality.hit10}% · MRR ${quality.mrr}`)
381
+ }
382
+ if (firstQueryHits.length) {
383
+ console.log(`\nsample top-1 for "${queries[0]}" → ${firstQueryHits[0].entry.sourcePath || firstQueryHits[0].entry.id}`)
384
+ }
385
+ console.log(opts.keep ? `\ntemp stores kept at ${tmpStore}` : `\ntemp stores removed (${tmpStore})`)
386
+ }
387
+
388
+ if (!opts.keep) {
389
+ rmSync(tmpStore, { recursive: true, force: true })
390
+ rmSync(coldDir, { recursive: true, force: true })
391
+ }
392
+
393
+ async function readFileSafe(p) {
394
+ const { readFile } = await import('node:fs/promises')
395
+ return readFile(p, 'utf8')
396
+ }