@a9i5k4/dsh-auto-memory 3.1.0 → 3.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/README.zh-CN.md +1 -1
- package/docs/CONTRIBUTORS.html +2 -2
- package/docs/HANDBOOK.md +13 -9
- package/lib/client.js +786 -134
- package/lib/config-io.js +59 -6
- package/lib/context-host.js +10 -1
- package/lib/evidence-store.js +27 -1
- package/lib/index.js +937 -51
- package/lib/jsonl-tail-cursor.js +75 -0
- package/lib/m7-index-sync-host.js +4 -6
- package/lib/memory-hub.js +25 -2
- package/lib/migrate-pack.js +431 -0
- package/lib/python-setup.js +109 -27
- package/lib/recall-stats.js +242 -0
- package/lib/rules-layer.js +106 -15
- package/lib/semantic-js.js +25 -7
- package/lib/shadow-host.js +41 -2
- package/lib/subagent-gc.js +5 -1
- package/lib/wb-sidecar.js +8 -3
- package/package.json +1 -1
package/lib/python-setup.js
CHANGED
|
@@ -20,7 +20,9 @@ import { promisify } from 'node:util'
|
|
|
20
20
|
import { createHash } from 'node:crypto'
|
|
21
21
|
import { homedir } from 'node:os'
|
|
22
22
|
import path from 'node:path'
|
|
23
|
-
import { existsSync, mkdirSync, writeFileSync, readFileSync, statSync, createWriteStream } from 'node:fs'
|
|
23
|
+
import { existsSync, mkdirSync, writeFileSync, readFileSync, statSync, createWriteStream, rmSync } from 'node:fs'
|
|
24
|
+
import { Readable } from 'node:stream'
|
|
25
|
+
import { pipeline } from 'node:stream/promises'
|
|
24
26
|
import { readFile, writeFile, rm } from 'node:fs/promises'
|
|
25
27
|
|
|
26
28
|
const execFileP = promisify(execFile)
|
|
@@ -266,32 +268,112 @@ function configReadyForModels() {
|
|
|
266
268
|
return snapshot()
|
|
267
269
|
}
|
|
268
270
|
|
|
269
|
-
/**
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
271
|
+
/**
|
|
272
|
+
* 断点续传下载(issue #105 重写)。旧实现有三处各自足以致命的点:
|
|
273
|
+
*
|
|
274
|
+
* 1. 对 `fetch` 返回的 **WHATWG** `ReadableStream` 调 `.on('data')` / `.destroy()`(那是 Node 流的
|
|
275
|
+
* API)⇒ 同步 TypeError 被外层 `.catch(reject)` 吞成一句「下载失败」。**自 v2.1.5 起本档下载从未成功。**
|
|
276
|
+
* 2. 回调里只累加 `done`,**全程没有 `stream.write()`** ⇒ 即便修好 (1),目标文件也必然 0 字节。
|
|
277
|
+
* 3. 无条件 `flags:'a'` 往 `.part` 上追加:既不校验这块 `.part` 属于哪个 URL,也不看服务端是否
|
|
278
|
+
* 真按 Range 响应 ⇒ 换镜像或服务端忽略 Range 回全量时,两次不同的响应体被拼成同一个文件。
|
|
279
|
+
*
|
|
280
|
+
* 现:`Readable.fromWeb` + `pipeline` 真写入;`.part.meta.json` 记归属(url + 已取字节);未按 206
|
|
281
|
+
* 续传就丢弃重来;取消走 AbortController(同时传给 `fetch`,断的是连接而不只是本地流)。
|
|
282
|
+
*/
|
|
283
|
+
async function downloadWithResume(url, target, onProgress, isCancelled) {
|
|
284
|
+
const part = target + '.part'
|
|
285
|
+
const metaPath = part + '.meta.json'
|
|
286
|
+
const bytesOnDisk = () => { try { return existsSync(part) ? statSync(part).size : 0 } catch (_) { return 0 } }
|
|
287
|
+
const saveMeta = (bytes) => { try { writeFileSync(metaPath, JSON.stringify({ url, bytes, at: Date.now() }), 'utf8') } catch (_) {} }
|
|
288
|
+
const dropPart = () => { try { rmSync(part, { force: true }) } catch (_) {} try { rmSync(metaPath, { force: true }) } catch (_) {} }
|
|
289
|
+
|
|
290
|
+
let done = bytesOnDisk()
|
|
291
|
+
if (done > 0) {
|
|
292
|
+
let prevMeta = null
|
|
293
|
+
try { prevMeta = JSON.parse(readFileSync(metaPath, 'utf8')) } catch (_) {}
|
|
294
|
+
const mine = !!prevMeta && prevMeta.url === url && Number(prevMeta.bytes) === done
|
|
295
|
+
if (!mine) {
|
|
296
|
+
diagOf('[降级] python-setup: .part 不归属本 URL(或缺归属记录),丢弃 ' + done + ' 字节重下 —— ' + path.basename(target))
|
|
297
|
+
dropPart()
|
|
298
|
+
done = 0
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
const ac = new AbortController()
|
|
302
|
+
let aborted = false
|
|
303
|
+
dlAbort = () => { aborted = true; try { ac.abort() } catch (_) {} }
|
|
304
|
+
try {
|
|
305
|
+
let resp = await fetch(url, { headers: done > 0 ? { Range: 'bytes=' + done + '-' } : {}, signal: ac.signal })
|
|
306
|
+
if (done > 0 && resp.status !== 206) {
|
|
307
|
+
// 服务端没理 Range(多半回 200 全量)⇒ 追加必然拼出坏文件。丢弃重来,不赌它的内容。
|
|
308
|
+
diagOf('[降级] python-setup: 服务端未按 Range 响应(status=' + resp.status + '),丢弃 ' + done + ' 字节 .part 重下')
|
|
309
|
+
dropPart()
|
|
310
|
+
done = 0
|
|
311
|
+
resp = await fetch(url, { headers: {}, signal: ac.signal })
|
|
312
|
+
}
|
|
313
|
+
if (!resp.ok && resp.status !== 206) throw new Error('HTTP ' + resp.status)
|
|
314
|
+
const resumeFrom = done
|
|
315
|
+
const total = Number(resp.headers.get('content-length') || 0) + resumeFrom
|
|
316
|
+
let lastTick = 0
|
|
317
|
+
const src = Readable.fromWeb(resp.body)
|
|
318
|
+
src.on('data', (chunk) => {
|
|
319
|
+
done += chunk.length
|
|
320
|
+
const now = Date.now()
|
|
321
|
+
if (now - lastTick > 500) { lastTick = now; onProgress(done, total) }
|
|
322
|
+
if (!aborted && typeof isCancelled === 'function' && isCancelled()) { aborted = true; try { ac.abort() } catch (_) {} }
|
|
323
|
+
})
|
|
324
|
+
await pipeline(src, createWriteStream(part, { flags: resumeFrom > 0 ? 'a' : 'w' }))
|
|
325
|
+
onProgress(done, total)
|
|
326
|
+
try { rmSync(metaPath, { force: true }) } catch (_) {}
|
|
327
|
+
await writeFileSyncSafe(part, target)
|
|
328
|
+
return done
|
|
329
|
+
} catch (e) {
|
|
330
|
+
if (aborted || (e && e.name === 'AbortError')) {
|
|
331
|
+
saveMeta(bytesOnDisk()) // 取消:保留 .part 与归属,下次可续传
|
|
332
|
+
const err = new Error('已取消'); err.cancelled = true
|
|
333
|
+
throw err
|
|
334
|
+
}
|
|
335
|
+
saveMeta(bytesOnDisk()) // 网络/写盘中断:同样记下归属,避免下次把半截体当可续传块盲拼
|
|
336
|
+
const err = new Error('下载失败(' + ((e && (e.name || 'Error')) || 'Error') + '): ' + String((e && e.message) || e).slice(0, 160))
|
|
337
|
+
err.cause = e
|
|
338
|
+
throw err
|
|
339
|
+
} finally {
|
|
340
|
+
dlAbort = null
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
/**
|
|
345
|
+
* 完整性校验(issue #105)。声明了 sha256 才真验,且**回读落盘文件**而非累计网络流——
|
|
346
|
+
* `lib/semantic-js.js:455-459` 记过同族事故:对网络流累积哈希 ⇒ 拼接出来的坏文件照样通过校验。
|
|
347
|
+
* 未声明 sha256 时不谎称验过:返回 `'size-only'`,由调用方打 `[降级]` 并如实暴露到状态面。
|
|
348
|
+
*/
|
|
349
|
+
async function verifyArtifact(file, spec = {}) {
|
|
350
|
+
// 校验失败单独打标:调用方据此**清除正式产物**(见 downloadModel 的 catch)。
|
|
351
|
+
const bad = (msg) => { const err = new Error(msg); err.integrityFailure = true; return err }
|
|
352
|
+
const size = statSync(file).size
|
|
353
|
+
const expected = Number(spec.bytes)
|
|
354
|
+
if (expected > 0) {
|
|
355
|
+
if (size !== expected) throw bad('大小不符(' + size + ' ≠ ' + expected + ' bytes): ' + path.basename(file))
|
|
356
|
+
} else if (spec.minBytes > 0 && size < spec.minBytes) {
|
|
357
|
+
throw bad('下载不完整(' + size + ' < ' + spec.minBytes + ' bytes): ' + path.basename(file))
|
|
358
|
+
}
|
|
359
|
+
const want = typeof spec.sha256 === 'string' ? spec.sha256 : ''
|
|
360
|
+
if (want.length > 0) {
|
|
361
|
+
const h = createHash('sha256')
|
|
362
|
+
await new Promise((res, rej) => {
|
|
363
|
+
const s = createReadStream(file)
|
|
364
|
+
s.on('data', (c) => h.update(c))
|
|
365
|
+
s.on('error', rej)
|
|
366
|
+
s.on('end', res)
|
|
367
|
+
})
|
|
368
|
+
const got = h.digest('hex')
|
|
369
|
+
if (got !== want) throw bad('sha256 不符(期望 ' + want.slice(0, 12) + '…, 实得 ' + got.slice(0, 12) + '…): ' + path.basename(file))
|
|
370
|
+
return 'sha256'
|
|
371
|
+
}
|
|
372
|
+
if (spec.kind === 'json') {
|
|
373
|
+
try { JSON.parse(readFileSync(file, 'utf8')) } catch (_) { throw bad('JSON 解析失败: ' + path.basename(file)) }
|
|
374
|
+
return 'size+json'
|
|
375
|
+
}
|
|
376
|
+
return 'size-only'
|
|
295
377
|
}
|
|
296
378
|
|
|
297
379
|
async function writeFileSyncSafe(from, to) {
|
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 召回统计(recall statistics)—— 只**记录**、不参与排序。
|
|
3
|
+
*
|
|
4
|
+
* **为什么只记录**(用户 2026-09-22 拍板的分步走):
|
|
5
|
+
* 加权是**会自我强化**的机制(召回越多 → 权重越高 → 越容易被召回)。在埋点口径未被真实数据
|
|
6
|
+
* 检验之前就加权,会把口径错误放大成系统性偏差。所以本模块**只累积计数并落盘**,
|
|
7
|
+
* `recall()` 的排序与返回**逐字节不变**(守卫断言:本模块的调用不改任何排序输入)。
|
|
8
|
+
*
|
|
9
|
+
* ★★ 三条通路**分账**,不混为一谈(用户 2026-09-22 明确要求):
|
|
10
|
+
* 同一个「召回」在三条通路里的含义完全不同,合成一个数会误导判断:
|
|
11
|
+
* | channel | 含义 | 语义 | 用来判断什么 |
|
|
12
|
+
* |----------|------------------|----------------------|------------------------|
|
|
13
|
+
* | `model` | 模型主动检索 | 「我明确想找」= 高信号 | 什么内容真的有用、被需要 |
|
|
14
|
+
* | `inject` | 每轮自动注入 | 「系统一直塞给我」= 成本 | 注入预算花在哪、哪段最费 |
|
|
15
|
+
* | `shadow` | 主动唤起(影子) | 「系统觉得该唤起」= 系统判断质量 | 唤起准不准、命中率如何 |
|
|
16
|
+
* ⇒ 若把 inject 与 model 合并,会把「模型懒得检索」误读成「这条不重要」,
|
|
17
|
+
* 而这个误读会被写进加权公式,越滚越偏。故**必须分开存、分开看**。
|
|
18
|
+
*
|
|
19
|
+
* 存储:`~/.dsh/memory/recall-stats.json`(单文件、原子写 tmp→rename、有上限裁剪)。
|
|
20
|
+
* 设计约束:零依赖(只用 node: 内置)+ fail-soft(统计坏了绝不影响召回本身)。
|
|
21
|
+
*/
|
|
22
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync, renameSync, rmSync } from 'node:fs'
|
|
23
|
+
import path from 'node:path'
|
|
24
|
+
|
|
25
|
+
export const RECALL_STATS_VERSION_PRE = 'recall_stats_v1'
|
|
26
|
+
|
|
27
|
+
/** 三条通路的稳定 id(写盘/前端都按它索引,禁止改拼写)。 */
|
|
28
|
+
export const RECALL_CHANNELS_PRE = ['model', 'inject', 'shadow']
|
|
29
|
+
|
|
30
|
+
/** 单条记忆的计数上限(超出按 count 升序淘汰最冷门的,防止文件无限增长)。 */
|
|
31
|
+
const MAX_ENTRIES = 2000
|
|
32
|
+
/** 保留天数(按天分布只留最近 N 天)。 */
|
|
33
|
+
const MAX_DAYS = 90
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* @param {{file: () => string, now?: () => number}} opts
|
|
37
|
+
* `file()` 每次调用都重新解析路径(DSH_HOME 可能变),必须是函数而非字符串。
|
|
38
|
+
*/
|
|
39
|
+
export function createRecallStatsPre(opts = {}) {
|
|
40
|
+
const nowFn = typeof opts.now === 'function' ? opts.now : () => Date.now()
|
|
41
|
+
const fileOf = typeof opts.file === 'function' ? opts.file : () => ''
|
|
42
|
+
let state = null
|
|
43
|
+
let dirty = false
|
|
44
|
+
|
|
45
|
+
const dayOf = (ms) => {
|
|
46
|
+
const d = new Date(ms)
|
|
47
|
+
const p = (n) => String(n).padStart(2, '0')
|
|
48
|
+
return `${d.getFullYear()}-${p(d.getMonth() + 1)}-${p(d.getDate())}`
|
|
49
|
+
}
|
|
50
|
+
/** 一条通路的空白账本。三条各自独立,互不汇总(汇总只发生在 snapshot 的展示层)。 */
|
|
51
|
+
const emptyChannel = () => ({
|
|
52
|
+
events: 0, // 这条通路发生了多少次(检索次数 / 注入轮数 / 唤起次数)
|
|
53
|
+
hits: 0, // 命中条目总数(inject 通道为「注入的条目数」)
|
|
54
|
+
zeroHit: 0, // 零命中次数(model 通道专有语义;其它通道记为 0)
|
|
55
|
+
ids: {}, // id → { count, first, last, layer, score, reason }
|
|
56
|
+
byLayer: {}, // layer → count
|
|
57
|
+
byDay: {}, // yyyy-mm-dd → count
|
|
58
|
+
lastAt: 0,
|
|
59
|
+
})
|
|
60
|
+
const empty = () => ({
|
|
61
|
+
version: RECALL_STATS_VERSION_PRE,
|
|
62
|
+
since: nowFn(),
|
|
63
|
+
channels: { model: emptyChannel(), inject: emptyChannel(), shadow: emptyChannel() },
|
|
64
|
+
})
|
|
65
|
+
|
|
66
|
+
/** 兼容旧结构(v1 早期是扁平 queries/hits/byLayer…)→ 迁到 channels.model,不丢数据。 */
|
|
67
|
+
function migrate(d) {
|
|
68
|
+
if (d && d.channels && typeof d.channels === 'object') {
|
|
69
|
+
const base = empty()
|
|
70
|
+
for (const ch of RECALL_CHANNELS_PRE) {
|
|
71
|
+
base.channels[ch] = Object.assign(emptyChannel(), (d.channels && d.channels[ch]) || {})
|
|
72
|
+
}
|
|
73
|
+
base.since = d.since || base.since
|
|
74
|
+
return base
|
|
75
|
+
}
|
|
76
|
+
const base = empty()
|
|
77
|
+
const m = base.channels.model
|
|
78
|
+
m.events = Number(d.queries) || 0
|
|
79
|
+
m.zeroHit = Number(d.zeroHit) || 0
|
|
80
|
+
m.ids = (d.hits && typeof d.hits === 'object') ? d.hits : {}
|
|
81
|
+
m.byLayer = (d.byLayer && typeof d.byLayer === 'object') ? d.byLayer : {}
|
|
82
|
+
m.byDay = (d.byDay && typeof d.byDay === 'object') ? d.byDay : {}
|
|
83
|
+
m.hits = Object.keys(m.ids).reduce((n, k) => n + ((m.ids[k] && m.ids[k].count) || 0), 0)
|
|
84
|
+
m.lastAt = Number(d.lastQueryAt) || 0
|
|
85
|
+
base.since = d.since || base.since
|
|
86
|
+
return base
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
function load() {
|
|
90
|
+
if (state) return state
|
|
91
|
+
state = empty()
|
|
92
|
+
try {
|
|
93
|
+
const f = fileOf()
|
|
94
|
+
if (f && existsSync(f)) {
|
|
95
|
+
const d = JSON.parse(readFileSync(f, 'utf8'))
|
|
96
|
+
if (d && typeof d === 'object' && d.version === RECALL_STATS_VERSION_PRE) {
|
|
97
|
+
state = migrate(d)
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
} catch (_) { state = empty() } // 坏文件不阻塞:从零开始,下一轮覆盖写回
|
|
101
|
+
return state
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
function save() {
|
|
105
|
+
if (!dirty) return false
|
|
106
|
+
try {
|
|
107
|
+
const f = fileOf()
|
|
108
|
+
if (!f) return false
|
|
109
|
+
mkdirSync(path.dirname(f), { recursive: true })
|
|
110
|
+
const tmp = f + '.tmp'
|
|
111
|
+
writeFileSync(tmp, JSON.stringify(state), 'utf8')
|
|
112
|
+
renameSync(tmp, f) // 同目录原子替换
|
|
113
|
+
dirty = false
|
|
114
|
+
return true
|
|
115
|
+
} catch (_) {
|
|
116
|
+
try { if (fileOf()) rmSync(fileOf() + '.tmp', { force: true }) } catch (_) {}
|
|
117
|
+
return false
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/** 裁剪:**每通道**条目数上限 + 天数上限。按 count 升序淘汰最冷门的。 */
|
|
122
|
+
function prune() {
|
|
123
|
+
for (const ch of RECALL_CHANNELS_PRE) {
|
|
124
|
+
const c = state.channels[ch]
|
|
125
|
+
if (!c) continue
|
|
126
|
+
const ids = Object.keys(c.ids)
|
|
127
|
+
if (ids.length > MAX_ENTRIES) {
|
|
128
|
+
ids.sort((a, b) => (c.ids[a].count || 0) - (c.ids[b].count || 0))
|
|
129
|
+
for (const id of ids.slice(0, ids.length - MAX_ENTRIES)) delete c.ids[id]
|
|
130
|
+
}
|
|
131
|
+
const days = Object.keys(c.byDay)
|
|
132
|
+
if (days.length > MAX_DAYS) {
|
|
133
|
+
days.sort()
|
|
134
|
+
for (const d of days.slice(0, days.length - MAX_DAYS)) delete c.byDay[d]
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* 记一次事件。**只读入参、绝不修改** —— 调用方传的 hits/顺序不因本调用而变。
|
|
141
|
+
*
|
|
142
|
+
* ★★ channel 必须显式给:`model`(模型主动检索)/ `inject`(每轮自动注入)/ `shadow`(主动唤起)。
|
|
143
|
+
* 缺省回落到 `model` 只是为了兼容旧调用点;**新调用点一律显式传**。
|
|
144
|
+
*
|
|
145
|
+
* @param {Array<{id:string, layer?:string, score?:number, reason?:string}>} hits
|
|
146
|
+
* @param {{channel?:'model'|'inject'|'shadow', zero?:boolean, label?:string, chars?:number}} [meta]
|
|
147
|
+
*/
|
|
148
|
+
function observe(hits, meta = {}) {
|
|
149
|
+
try {
|
|
150
|
+
const s = load()
|
|
151
|
+
const t = nowFn()
|
|
152
|
+
const ch = RECALL_CHANNELS_PRE.indexOf(meta.channel) >= 0 ? meta.channel : 'model'
|
|
153
|
+
const c = s.channels[ch]
|
|
154
|
+
c.events += 1
|
|
155
|
+
c.lastAt = t
|
|
156
|
+
const list = Array.isArray(hits) ? hits : []
|
|
157
|
+
// 零命中:对 model 通道是「检索了但没东西」(重要信号);其它通道同样计数,语义由展示层解释
|
|
158
|
+
if (!list.length) c.zeroHit += 1
|
|
159
|
+
c.hits += list.length
|
|
160
|
+
const d = dayOf(t)
|
|
161
|
+
for (const h of list) {
|
|
162
|
+
const id = h && h.id ? String(h.id) : ''
|
|
163
|
+
if (!id) continue
|
|
164
|
+
const cur = c.ids[id] || { count: 0, first: t, last: t, layer: '', score: 0, reason: '' }
|
|
165
|
+
cur.count += 1
|
|
166
|
+
cur.last = t
|
|
167
|
+
if (h.layer) cur.layer = String(h.layer)
|
|
168
|
+
if (typeof h.score === 'number' && isFinite(h.score)) cur.score = h.score
|
|
169
|
+
// label / chars:注入侧用来回答「哪一段最占预算」(无 id 的段落用它记账)
|
|
170
|
+
if (h.__label) cur.reason = String(h.__label)
|
|
171
|
+
if (typeof h.__chars === 'number' && isFinite(h.__chars)) cur.score = h.__chars
|
|
172
|
+
if (h.reason) cur.reason = String(h.reason)
|
|
173
|
+
c.ids[id] = cur
|
|
174
|
+
const lyr = String(h.layer || 'unknown')
|
|
175
|
+
c.byLayer[lyr] = (c.byLayer[lyr] || 0) + 1
|
|
176
|
+
c.byDay[d] = (c.byDay[d] || 0) + 1
|
|
177
|
+
}
|
|
178
|
+
prune()
|
|
179
|
+
dirty = true
|
|
180
|
+
return true
|
|
181
|
+
} catch (_) { return false } // fail-soft:统计绝不影响召回
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/** 把一条通路的账本转成只读视图(已排序,直接可画图)。 */
|
|
185
|
+
function viewOf(c) {
|
|
186
|
+
const items = Object.keys(c.ids).map((id) => Object.assign({ id }, c.ids[id]))
|
|
187
|
+
const top = items.slice().sort((a, b) => b.count - a.count || (b.last || 0) - (a.last || 0))
|
|
188
|
+
const byLayer = Object.keys(c.byLayer).map((k) => ({ key: k, count: c.byLayer[k] }))
|
|
189
|
+
.sort((a, b) => b.count - a.count)
|
|
190
|
+
const byDay = Object.keys(c.byDay).sort().map((k) => ({ day: k, count: c.byDay[k] }))
|
|
191
|
+
return {
|
|
192
|
+
events: c.events,
|
|
193
|
+
hits: c.hits,
|
|
194
|
+
zeroHit: c.zeroHit,
|
|
195
|
+
lastAt: c.lastAt,
|
|
196
|
+
distinct: items.length,
|
|
197
|
+
top: top.slice(0, 200),
|
|
198
|
+
byLayer,
|
|
199
|
+
byDay,
|
|
200
|
+
warm: top.filter((it) => it.count <= 1).length, // 只出现过一次的("冷门"候选)
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/** 面板/路由用的只读视图。**按通道分组返回**,不做跨通道汇总(汇总口径会误导)。 */
|
|
205
|
+
function snapshot() {
|
|
206
|
+
const s = load()
|
|
207
|
+
const channels = {}
|
|
208
|
+
for (const ch of RECALL_CHANNELS_PRE) channels[ch] = viewOf(s.channels[ch])
|
|
209
|
+
return {
|
|
210
|
+
version: RECALL_STATS_VERSION_PRE,
|
|
211
|
+
since: s.since,
|
|
212
|
+
channelIds: RECALL_CHANNELS_PRE.slice(),
|
|
213
|
+
channels,
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
/** 复位(面板按钮 / 测试)。 */
|
|
218
|
+
function reset() {
|
|
219
|
+
state = empty()
|
|
220
|
+
dirty = true
|
|
221
|
+
return save()
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* 注入侧的段级记账(没有条目 id 的段落也能记)。
|
|
226
|
+
* 用途:回答「哪一段最占注入预算」—— 段名当 id,字符数当 score。
|
|
227
|
+
* @param {Array<{name:string, chars:number, layer?:string}>} segs
|
|
228
|
+
*/
|
|
229
|
+
function observeInjection(segs) {
|
|
230
|
+
const list = (Array.isArray(segs) ? segs : [])
|
|
231
|
+
.filter((s) => s && s.name)
|
|
232
|
+
.map((s) => ({
|
|
233
|
+
id: 'seg:' + String(s.name),
|
|
234
|
+
layer: s.layer || 'section',
|
|
235
|
+
score: Number(s.chars) || 0,
|
|
236
|
+
__label: String(s.name) + ' · ' + (Number(s.chars) || 0) + ' 字符',
|
|
237
|
+
}))
|
|
238
|
+
return observe(list, { channel: 'inject' })
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
return { observe, observeInjection, snapshot, reset, save, _load: load, _state: () => state }
|
|
242
|
+
}
|
package/lib/rules-layer.js
CHANGED
|
@@ -28,7 +28,14 @@
|
|
|
28
28
|
* "模型**真的收到了**"取决于宿主最终请求 messages 的确认(`MASTER-PLAN-3.0.md §7` 的 **U6**,
|
|
29
29
|
* 尚未具备)⇒ **不得**据此宣称"规则的遵守问题已解决"(T7-7 的纪律)。
|
|
30
30
|
*
|
|
31
|
-
* S9 合规:零 IO
|
|
31
|
+
* S9 合规:零 IO、纯函数、无网络/无 LLM/无子进程/无 await。UTF-8 无 BOM。
|
|
32
|
+
*
|
|
33
|
+
* **P9 更新(2026-09-22)**:规则段从此**认状态** —— superseded / retracted 的条目进 retired
|
|
34
|
+
* 分项、**不再注入**(旧行为是规则段无视状态、过时条目每轮照发)。
|
|
35
|
+
* ★**本模块的「零 import」不变式保持不变**:状态解析用**等价内联**(statusOfEntryPre /
|
|
36
|
+
* stripStatusLinesPre,语法与 note-status.js 一致),不引入模块依赖。
|
|
37
|
+
* 理由:p6b 与 p6a 两条守卫都锁这条不变式,且 l0-extract.js:6504-6508 有同样先例;
|
|
38
|
+
* 等价性由守卫 tests/smoke/smoke-test-p9-rules-lifecycle-pre.mjs 逐例对照断言。
|
|
32
39
|
*/
|
|
33
40
|
|
|
34
41
|
export const RULES_LAYER_VERSION = 'rules_layer_v1'
|
|
@@ -70,11 +77,73 @@ export const RULE_DESCRIPTION_MARKERS_V1 = Object.freeze([
|
|
|
70
77
|
/** 条目切分锚点:与 `l0-extract.js` 的记忆锚点完全一致(不另立一套)。 */
|
|
71
78
|
const MEM_ANCHOR_LINE_RE = /^<!--\s*memory:(mem_[0-9a-f]{32})\s*-->$/
|
|
72
79
|
|
|
73
|
-
/**
|
|
74
|
-
|
|
80
|
+
/**
|
|
81
|
+
* 条目内的日期小节 —— 既有记忆文件的实际结构。
|
|
82
|
+
*
|
|
83
|
+
* P9 修正:从「只认 ## + 纯日期」放宽到「任意级别 + 可带标题」——
|
|
84
|
+
* 旧形态漏掉 `## 2026-09-22 · <标题>` / `### 2026-09-18 · <标题>` 这类带标题的日期小节。
|
|
85
|
+
*/
|
|
86
|
+
const DATE_SECTION_RE = /^#{1,6}\s*\d{4}-\d{2}-\d{2}\s*$/
|
|
87
|
+
|
|
88
|
+
/** ATX 标题行(任意级别)—— 小节标记,不是规则正文。 */
|
|
89
|
+
const ATX_HEADING_RE = /^#{1,6}\s+(\S.*)$/
|
|
90
|
+
|
|
91
|
+
/** 标题行的可读文本(剥 # 前缀)—— 正文缺失时的兜底,绝不留空条。 */
|
|
92
|
+
function headingTextPre(line) {
|
|
93
|
+
const m = ATX_HEADING_RE.exec(clean(line))
|
|
94
|
+
return m ? m[1].trim() : ''
|
|
95
|
+
}
|
|
75
96
|
|
|
76
97
|
const clean = (s) => String(s == null ? '' : s)
|
|
77
98
|
|
|
99
|
+
/**
|
|
100
|
+
* 条目状态行语法(与 note-status.js **完全一致**;刻意内联见文件头 P9 说明)。
|
|
101
|
+
* 只认整行、只认三态;非法 memoryId / 未知属性一律丢弃(fail-soft)。
|
|
102
|
+
*/
|
|
103
|
+
const STATUS_LINE_RE_V1 = /^<!--\s*dsh-status:\s*(current|superseded|retracted)\s*((?:\w+=(?:"[^"]*"|[^\s"]+)\s*)*)-->$/
|
|
104
|
+
const STATUS_ATTR_RE_V1 = /(\w+)=(?:"([^"]*)"|([^\s"]+))/g
|
|
105
|
+
const MEM_ID_RE_V1 = /^mem_[0-9a-f]{32}$/
|
|
106
|
+
const STATUS_REASON_MAX_V1 = 120
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* 从条目正文解析状态:**最后一个**状态行胜出(口径同 note-status.js 的 statusOfBodyPre)。
|
|
110
|
+
* 无状态行 ⇒ { status: current }。等价性由 tests/smoke/smoke-test-p9-rules-lifecycle-pre.mjs 逐例对照。
|
|
111
|
+
* @param {string} body 条目正文(不含锚点行)
|
|
112
|
+
* @returns {{status:string, supersededBy?:string, reason?:string}}
|
|
113
|
+
*/
|
|
114
|
+
export function statusOfEntryPre(body) {
|
|
115
|
+
const lines = clean(body).split(/\r?\n/)
|
|
116
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
117
|
+
const m = STATUS_LINE_RE_V1.exec(lines[i].trim())
|
|
118
|
+
if (!m) continue
|
|
119
|
+
const out = { status: m[1] }
|
|
120
|
+
STATUS_ATTR_RE_V1.lastIndex = 0
|
|
121
|
+
let a
|
|
122
|
+
while ((a = STATUS_ATTR_RE_V1.exec(m[2] || '')) !== null) {
|
|
123
|
+
const k = a[1]
|
|
124
|
+
const v = a[2] !== undefined ? a[2] : a[3]
|
|
125
|
+
if (k === 'by') { if (MEM_ID_RE_V1.test(String(v || ''))) out.supersededBy = String(v) }
|
|
126
|
+
else if (k === 'reason') {
|
|
127
|
+
const r = String(v || '').replace(/[\r\n]+/g, ' ').trim().slice(0, STATUS_REASON_MAX_V1)
|
|
128
|
+
if (r) out.reason = r
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
return out
|
|
132
|
+
}
|
|
133
|
+
return { status: 'current' }
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* 剥掉**整行**状态行,其余行逐字保留(口径同 note-status.js 的 stripStatusLinePre)。
|
|
138
|
+
* @param {string} body
|
|
139
|
+
* @returns {string}
|
|
140
|
+
*/
|
|
141
|
+
export function stripStatusLinesPre(body) {
|
|
142
|
+
const s = clean(body)
|
|
143
|
+
if (!s.includes('<!-- dsh-status:')) return s
|
|
144
|
+
return s.split(/\r?\n/).filter((ln) => STATUS_LINE_RE_V1.exec(ln.trim()) === null).join('\n')
|
|
145
|
+
}
|
|
146
|
+
|
|
78
147
|
/**
|
|
79
148
|
* 把记忆文本按锚点切成条目(保持既有文件结构语义:无锚点的前置内容归入首条)。
|
|
80
149
|
*
|
|
@@ -143,8 +212,8 @@ export function extractRulesLayerPre(input = {}) {
|
|
|
143
212
|
const o = input && typeof input === 'object' ? input : {}
|
|
144
213
|
const mode = clampMode(o.rulesLayeringMode)
|
|
145
214
|
const empty = {
|
|
146
|
-
version: RULES_LAYER_VERSION, mode, enabled: false, text: '', rules: [], candidates: [], references: [],
|
|
147
|
-
counts: { entries: 0, rules: 0, candidates: 0, references: 0 }, chars: { rules: 0, candidates: 0, total: 0 },
|
|
215
|
+
version: RULES_LAYER_VERSION, mode, enabled: false, text: '', rules: [], candidates: [], references: [], retired: [],
|
|
216
|
+
counts: { entries: 0, rules: 0, candidates: 0, references: 0, retired: 0 }, chars: { rules: 0, candidates: 0, total: 0 },
|
|
148
217
|
}
|
|
149
218
|
if (mode === null) return empty
|
|
150
219
|
|
|
@@ -161,22 +230,32 @@ export function extractRulesLayerPre(input = {}) {
|
|
|
161
230
|
const rules = []
|
|
162
231
|
const candidates = []
|
|
163
232
|
const references = []
|
|
233
|
+
const retired = []
|
|
164
234
|
for (const e of sources) {
|
|
165
|
-
|
|
166
|
-
const
|
|
235
|
+
// ★P9:先读状态、再分类 —— 状态行是元数据不是正文(留着会被当规则语汇参与分类)
|
|
236
|
+
const st = statusOfEntryPre(e.text)
|
|
237
|
+
const content = stripStatusLinesPre(e.text)
|
|
238
|
+
if (st.status !== 'current') {
|
|
239
|
+
retired.push({
|
|
240
|
+
layer: e.layer, id: e.id, anchorLine: e.anchorLine,
|
|
241
|
+
status: st.status, supersededBy: st.supersededBy || '', reason: st.reason || '',
|
|
242
|
+
text: content,
|
|
243
|
+
})
|
|
244
|
+
continue
|
|
245
|
+
}
|
|
246
|
+
const c = classifyText(content)
|
|
247
|
+
const rec = { layer: e.layer, id: e.id, anchorLine: e.anchorLine, kind: c.kind, confidence: c.confidence, reasons: c.reasons, text: content }
|
|
167
248
|
if (c.kind === 'rule') rules.push(rec)
|
|
168
249
|
else if (c.kind === 'candidate') candidates.push(rec)
|
|
169
250
|
else references.push(rec)
|
|
170
251
|
}
|
|
171
252
|
|
|
172
253
|
// 规则层文本:只含规则条目(高/中置信),逐条一行摘要,**不掺参考类**
|
|
173
|
-
const
|
|
174
|
-
const
|
|
175
|
-
if (rules.length) parts.push(rules.map(renderLine).join('\n'))
|
|
176
|
-
const text = parts.join('\n')
|
|
254
|
+
const rowsOut = rules.map((r) => '- ' + ruleSummaryPre(r.text)).filter((s) => s !== '- ')
|
|
255
|
+
const text = rowsOut.join('\n')
|
|
177
256
|
return {
|
|
178
|
-
version: RULES_LAYER_VERSION, mode, enabled: true, text, rules, candidates, references,
|
|
179
|
-
counts: { entries: sources.length, rules: rules.length, candidates: candidates.length, references: references.length },
|
|
257
|
+
version: RULES_LAYER_VERSION, mode, enabled: true, text, rules, candidates, references, retired,
|
|
258
|
+
counts: { entries: sources.length, rules: rules.length, candidates: candidates.length, references: references.length, retired: retired.length },
|
|
180
259
|
chars: { rules: text.length, candidates: candidates.reduce((a, r) => a + ruleSummaryPre(r.text).length + 2, 0), total: text.length },
|
|
181
260
|
}
|
|
182
261
|
}
|
|
@@ -209,8 +288,16 @@ export function firstLine(text, max = 200) {
|
|
|
209
288
|
*/
|
|
210
289
|
export function ruleSummaryPre(text, max = 200) {
|
|
211
290
|
const lines = clean(text).split('\n').map((l) => l.trim()).filter(Boolean)
|
|
291
|
+
// ★P9(2026-09-22):ATX 标题一律视为**小节标记**,优先渲染正文首行;
|
|
292
|
+
// 正文缺失才回落到标题文本。旧行为把「## <日期> · <标题>」当正文渲染 ⇒
|
|
293
|
+
// 规则段里进去的全是标题,规则本体一条都不在场(实测 54 条规则多数如此)。
|
|
294
|
+
let headingFallback = ''
|
|
212
295
|
for (const l of lines) {
|
|
213
296
|
if (DATE_SECTION_RE.test(l)) continue
|
|
297
|
+
if (ATX_HEADING_RE.test(l)) {
|
|
298
|
+
if (!headingFallback) headingFallback = headingTextPre(l)
|
|
299
|
+
continue
|
|
300
|
+
}
|
|
214
301
|
if (/^<!--/.test(l)) continue
|
|
215
302
|
if (/^[-*+]\s*$/.test(l)) continue
|
|
216
303
|
// 去掉列表前缀再返回,保持一条一行
|
|
@@ -218,9 +305,10 @@ export function ruleSummaryPre(text, max = 200) {
|
|
|
218
305
|
// ★P6B:剥行内 kind 标记与前置时间戳(只影响渲染,不回写盘上原文)
|
|
219
306
|
body = body.replace(LOG_KIND_TAG_RE_V1, ' ').trim()
|
|
220
307
|
body = body.replace(/^\d{1,2}:\d{2}\s+/, '').trim()
|
|
308
|
+
if (!body) continue
|
|
221
309
|
return body.length > max ? body.slice(0, Math.max(1, max - 1)) + '…' : body
|
|
222
310
|
}
|
|
223
|
-
return ''
|
|
311
|
+
return headingFallback.length > max ? headingFallback.slice(0, Math.max(1, max - 1)) + '…' : headingFallback
|
|
224
312
|
}
|
|
225
313
|
|
|
226
314
|
/**
|
|
@@ -241,7 +329,10 @@ export function renderRulesSectionPre(rules, opts = {}) {
|
|
|
241
329
|
if (!list.length) return { text: '', chars: 0, guide: '' }
|
|
242
330
|
const title = clean((opts && opts.title) || RULES_SECTION_TITLE_V1)
|
|
243
331
|
const guide = clean((opts && opts.guide) || RULES_SECTION_GUIDE_V1)
|
|
244
|
-
|
|
332
|
+
// ★P9:空摘要**不渲染**(旧写法会产出裸 "- " 行;放宽小节标记后更易出现)
|
|
333
|
+
const rows = list.map((x) => '- ' + ruleSummaryPre(x.text)).filter((r) => r !== '- ')
|
|
334
|
+
if (!rows.length) return { text: '', chars: 0, guide: '' }
|
|
335
|
+
const text = '\n' + title + '\n' + guide + '\n' + rows.join('\n')
|
|
245
336
|
return { text, chars: text.length, guide }
|
|
246
337
|
}
|
|
247
338
|
|
package/lib/semantic-js.js
CHANGED
|
@@ -68,6 +68,22 @@ export function fuseD6Pre(pairs) {
|
|
|
68
68
|
|
|
69
69
|
const E5_MODELS_SUBDIR = 'multilingual-e5-small'
|
|
70
70
|
|
|
71
|
+
/**
|
|
72
|
+
* ★P3-14(2026-09-22)→ **P10-B 修正(同日晚)**:`artifacts/m7-live-pre/js-semantic-trial` 是**纯维护者机器路径**
|
|
73
|
+
* (`.gitignore` 已排除其 models/ 与 node_modules/,用户机上结构性不存在)。
|
|
74
|
+
* P3-14 原用「环境变量 `DAM_DEV_TREE=1` 显式开启」来排除发布包 —— 但那个开关**把维护者本机也一起排除了**
|
|
75
|
+
* (实测:开发机该目录有 112.8MB 模型 + peer,却因开关默认关而报 `setup-both`)。
|
|
76
|
+
* ⇒ 改为**存在性判定**:目录存在即视为开发树;用户机上该目录结构性不存在 ⇒ 发布包行为不变。
|
|
77
|
+
* `DAM_DEV_TREE=1` 保留为**显式强制开启**(离线调试且目录另置时用)。
|
|
78
|
+
*/
|
|
79
|
+
const DEV_TREE_ENABLED = process.env.DAM_DEV_TREE === '1'
|
|
80
|
+
|
|
81
|
+
/** 开发树 artifacts 候选根(显式开关**或该目录实际存在**时非 null —— 见上方 P10-B 修正)。 */
|
|
82
|
+
function devTreeRoot(pluginDir) {
|
|
83
|
+
const root = path.join(pluginDir, '..', 'artifacts', 'm7-live-pre', 'js-semantic-trial')
|
|
84
|
+
return (DEV_TREE_ENABLED || existsSync(root)) ? root : null
|
|
85
|
+
}
|
|
86
|
+
|
|
71
87
|
function defaultModelsDirCandidates(pluginDir) {
|
|
72
88
|
// 用户目录优先(#15 后续/B 修复):~/.dsh/models/js-semantic/ 跨插件升级存活——
|
|
73
89
|
// 包目录(lib/models)在 npm 更新时会被整体重装,下载的 130MB 模型曾被冲掉。
|
|
@@ -76,8 +92,7 @@ function defaultModelsDirCandidates(pluginDir) {
|
|
|
76
92
|
return [
|
|
77
93
|
path.join(dshHome, 'models', 'js-semantic'),
|
|
78
94
|
path.join(pluginDir, 'models'),
|
|
79
|
-
|
|
80
|
-
]
|
|
95
|
+
].concat(devTreeRoot(pluginDir) ? [path.join(devTreeRoot(pluginDir), 'models')] : [])
|
|
81
96
|
}
|
|
82
97
|
|
|
83
98
|
function defaultPeerDirCandidates(pluginDir) {
|
|
@@ -87,13 +102,12 @@ function defaultPeerDirCandidates(pluginDir) {
|
|
|
87
102
|
// 2) <pkg>/node_modules —— 包内邻接位(lib 上 1 级再进 node_modules)
|
|
88
103
|
// 3) lib 上 3 级直拼 @huggingface —— 标准布局即 <root>/node_modules(pnpm hoisted/npm 提升),
|
|
89
104
|
// pnpm isolated 下即虚拟存储包内位 .pnpm/<hash>/node_modules
|
|
90
|
-
// 4) 开发树 artifacts
|
|
105
|
+
// 4) 开发树 artifacts(★P3-14:仅 DAM_DEV_TREE=1 时纳入;该路径 .gitignore 排除、用户机恒不存在)
|
|
91
106
|
return [
|
|
92
107
|
path.join(pluginDir, 'node_modules', '@huggingface', 'transformers'),
|
|
93
108
|
path.join(pluginDir, '..', 'node_modules', '@huggingface', 'transformers'),
|
|
94
109
|
path.join(pluginDir, '..', '..', '..', '@huggingface', 'transformers'),
|
|
95
|
-
|
|
96
|
-
]
|
|
110
|
+
].concat(devTreeRoot(pluginDir) ? [path.join(devTreeRoot(pluginDir), 'node_modules', '@huggingface', 'transformers')] : [])
|
|
97
111
|
}
|
|
98
112
|
|
|
99
113
|
/**
|
|
@@ -180,8 +194,12 @@ export function probeJsSemanticAssets(pluginDir, extraDirs, degradedReason) {
|
|
|
180
194
|
const modelCands = [
|
|
181
195
|
path.join(dshHome, 'models', 'js-semantic', E5_MODELS_SUBDIR, 'onnx', 'model_quantized.onnx'),
|
|
182
196
|
path.join(pluginDir, 'models', E5_MODELS_SUBDIR, 'onnx', 'model_quantized.onnx'),
|
|
183
|
-
|
|
184
|
-
|
|
197
|
+
// ★P10-C(2026-09-22,用户裁定):dev 树闸门从「环境变量显式开启」统一为**存在性判定**。
|
|
198
|
+
// P10-B 只改了本文件另外两处(defaultModelsDirCandidates / defaultPeerDirCandidates),
|
|
199
|
+
// **探针这处漏改** ⇒ 开发机上 peer 已可解析(面板不再报 setup-both),但资产恒判 missing
|
|
200
|
+
// ⇒ 面板显示「JS 引擎模型 ✗ 未就绪」(用户报障)。三处口径现已一致。
|
|
201
|
+
// 保护意图不变:用户机上 `artifacts/m7-live-pre/` 结构性不存在 ⇒ 发布包行为零变化。
|
|
202
|
+
].concat(devTreeRoot(pluginDir) ? [path.join(devTreeRoot(pluginDir), 'models', E5_MODELS_SUBDIR, 'onnx', 'model_quantized.onnx')] : [])
|
|
185
203
|
const modelOnnx = modelCands.find((c) => existsSync(c)) || modelCands[0]
|
|
186
204
|
const assetPresent = existsSync(modelOnnx)
|
|
187
205
|
const peerDir = resolvePeerTransformersDir(pluginDir, extraDirs)
|