@a9i5k4/dsh-auto-memory 2.3.1 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,9 +4,11 @@
4
4
  * 四步链路(全部走本模块,UI 只管展示状态与点击):
5
5
  * ①detect — 探测系统 Python(≥3.9)/既有 venv/模型本体;全部只读。
6
6
  * ②venv — python -m venv <userDir>/python-engine/.venv(幂等:已存在直接跳过)。
7
- * ③deps — venv 内 pip 安装运行依赖(transformers/onnxruntime/torch,清华镜像兜底)。
7
+ * ③deps — venv 内 pip 安装运行依赖(transformers/onnxruntime,清华镜像兜底;int8 档不需要 torch)。
8
8
  * ④model — BGE-M3 int8(~539MB)下载到 <userDir>/python-engine/models/,
9
9
  * cn(hf-mirror)/intl(hf 官方)双通道+SHA256 校验,复用 JS 档下载器的状态机形态。
10
+ * ④' — 模型齐备后**必须回写 embedding-config.json**(provider/modelDir/onnxFile/dimension),
11
+ * 否则 worker load_embedder 抛 unknown embedding provider —— 这是"装了用不了"的头号断点。
10
12
  *
11
13
  * 设计约束:
12
14
  * - 一切落盘在用户目录(~/.dsh/python-engine/),npm 包目录升级会被覆盖,绝不放模型。
@@ -18,7 +20,7 @@ import { promisify } from 'node:util'
18
20
  import { createHash } from 'node:crypto'
19
21
  import { homedir } from 'node:os'
20
22
  import path from 'node:path'
21
- import { existsSync, mkdirSync, writeFileSync, statSync, createWriteStream } from 'node:fs'
23
+ import { existsSync, mkdirSync, writeFileSync, readFileSync, statSync, createWriteStream } from 'node:fs'
22
24
  import { readFile, writeFile, rm } from 'node:fs/promises'
23
25
 
24
26
  const execFileP = promisify(execFile)
@@ -45,6 +47,15 @@ const TOKENIZER_FILES = ['config.json', 'tokenizer.json', 'tokenizer_config.json
45
47
  const PIP_DEPS_CPU = ['transformers', 'onnxruntime']
46
48
  const PIP_DEPS_GPU_EXTRA = ['onnxruntime-gpu']
47
49
 
50
+ /** 探针要与安装清单**同口径**:int8 档 load_embedder 只 import numpy+onnxruntime+transformers;
51
+ * 历史上探针多要一个 torch,而 PIP_DEPS_CPU 从不装 torch → depsOk 恒 false,"全部就绪"永不出现。 */
52
+ const DEPS_PROBE = 'import transformers, onnxruntime, numpy; print("deps-ok")'
53
+
54
+ /** worker 侧 load_embedder 的档位名(m7_embedding_v1.PROVIDER_REAL_INT8);发布构建会把 -pre-v1 折成 -v1。 */
55
+ const PROVIDER_ID_INT8 = 'bge-m3-onnx-int8-v1'
56
+ /** dense 维度(BGE-M3 = 1024),写进 embedding-config.json 的 dimension。 */
57
+ const EMBED_DIMENSION = 1024
58
+
48
59
  export function createPythonSetupPre(opts = {}) {
49
60
  const dshHomeOf = typeof opts.dshHome === 'function' ? opts.dshHome : () => opts.dshHome || path.join(homedir(), '.dsh')
50
61
  const diagOf = typeof opts.diag === 'function' ? opts.diag : () => {}
@@ -64,6 +75,7 @@ export function createPythonSetupPre(opts = {}) {
64
75
  chosenPython: '',
65
76
  venvOk: false,
66
77
  depsOk: false,
78
+ configOk: false,
67
79
  dl: { bytesDone: 0, bytesTotal: MODEL_SPEC.bytes, mirror: '', startedAt: 0, etaSec: 0 },
68
80
  cancelled: false,
69
81
  }
@@ -86,6 +98,43 @@ export function createPythonSetupPre(opts = {}) {
86
98
  return m ? { major: Number(m[1]), minor: Number(m[2]) } : null
87
99
  }
88
100
 
101
+ // ---------- 引擎侧 embedding-config.json(D1/D2:向导产物必须落在 worker 真正读取的位置与键上) ----------
102
+ /** worker 读 <dsh-home>/memory/semantic/embedding-config.json(见 worker_semantic_v1.load_embedding_config_from_env);
103
+ * 发布构建会把 `semantic` 折成 `semantic`,故这里写带 -pre 的规范名。
104
+ * 历史缺陷:向导曾写 <dsh-home>/memory/semantic/(缺 -pre),pre 线 worker 从不读该路径 → 装完即"找不到配置"。 */
105
+ const embeddingConfigPath = () => path.join(dshHomeOf(), 'memory', 'semantic', 'embedding-config.json')
106
+
107
+ /** read-modify-write:只覆盖本次传入的键,保留引擎运行期自管的 search/activationPolicy/activationEmitMode 等。 */
108
+ async function patchEmbeddingConfig(patch) {
109
+ const cfgPath = embeddingConfigPath()
110
+ mkdirSync(path.dirname(cfgPath), { recursive: true })
111
+ let cfg = {}
112
+ try { cfg = JSON.parse(await readFile(cfgPath, 'utf8')) } catch (_) {}
113
+ if (!cfg || typeof cfg !== 'object' || Array.isArray(cfg)) cfg = {}
114
+ Object.assign(cfg, patch)
115
+ await writeFile(cfgPath, JSON.stringify(cfg, null, 2), 'utf8')
116
+ return cfgPath
117
+ }
118
+
119
+ /** 就绪口径(D4):配置声明的 provider/modelDir 必须指向磁盘上真实存在的 onnx + tokenizer;
120
+ * 只看"onnx 已下载"会给出假绿 —— worker 起来照样抛 unknown embedding provider / KeyError modelDir。
121
+ * 注意与下载清单 TOKENIZER_FILES 的区别:那是"下全"的清单(slow/fast 两条路径都覆盖),
122
+ * 这里判的是**能否加载** —— AutoTokenizer 默认 fast 路径有 tokenizer.json 即可,缺 sentencepiece.bpe.model 不算坏。 */
123
+ function configReadyForModels() {
124
+ try {
125
+ const cfg = JSON.parse(readFileSync(embeddingConfigPath(), 'utf8'))
126
+ if (!cfg || typeof cfg !== 'object' || Array.isArray(cfg)) return false
127
+ if (String(cfg.provider || '') !== PROVIDER_ID_INT8) return false
128
+ const base = String(cfg.modelDir || '').trim()
129
+ if (!base) return false
130
+ const rel = String(cfg.onnxFile || 'onnx/model_int8.onnx')
131
+ if (!existsSync(path.join(base, ...rel.split('/')))) return false
132
+ const hasTokenizer = ['tokenizer.json', 'sentencepiece.bpe.model'].some((f) => existsSync(path.join(base, f)))
133
+ if (!hasTokenizer) return false
134
+ return ['config.json', 'tokenizer_config.json'].every((f) => existsSync(path.join(base, f)))
135
+ } catch (_) { return false }
136
+ }
137
+
89
138
  /** ①环境探测:venv(推荐)→ 系统 python/python3/py launcher;版本 ≥3.9 且 <3.13(onnxruntime 兼容上界)。 */
90
139
  async function detect() {
91
140
  st.phase = 'detecting'; st.error = ''
@@ -106,7 +155,8 @@ export function createPythonSetupPre(opts = {}) {
106
155
  }
107
156
  // 既有成果快照
108
157
  st.venvOk = existsSync(venvPython())
109
- st.depsOk = st.venvOk ? (await probe(venvPython(), ['-c', 'import transformers, onnxruntime, torch; print("deps-ok")'])).ok : false
158
+ st.depsOk = st.venvOk ? (await probe(venvPython(), ['-c', DEPS_PROBE])).ok : false
159
+ st.configOk = configReadyForModels()
110
160
  // modelReady 判定(2026-09-09):onnx + tokenizer 5 件全齐才算就绪,避免 UI 显示 ✓ 但 sidecar 起不来
111
161
  st.modelReady = existsSync(modelPath()) && TOKENIZER_FILES.every((f) => existsSync(path.join(modelsDir(), f)))
112
162
  st.pythons = out
@@ -118,7 +168,7 @@ export function createPythonSetupPre(opts = {}) {
118
168
  return {
119
169
  phase: st.phase, error: st.error,
120
170
  pythons: st.pythons, chosenPython: st.chosenPython,
121
- venvOk: st.venvOk, depsOk: st.depsOk, modelReady: st.modelReady, wantGpu: !!st.wantGpu,
171
+ venvOk: st.venvOk, depsOk: st.depsOk, configOk: st.configOk, modelReady: st.modelReady, wantGpu: !!st.wantGpu,
122
172
  modelPath: modelPath(), venvPython: venvPython(),
123
173
  modelBytes: st.modelReady ? statSync(modelPath()).size : 0, modelExpectedBytes: MODEL_SPEC.bytes,
124
174
  dl: { ...st.dl },
@@ -163,16 +213,7 @@ export function createPythonSetupPre(opts = {}) {
163
213
  st.depsOk = !!r.ok
164
214
  if (!r.ok) { st.error = '依赖安装失败: ' + (r.tail || ''); st.phase = 'error'; return snapshot() }
165
215
  // GPU 偏好写入 embedding-config.json(worker 读 config['gpu'] 选 CUDA/CPU provider;read-modify-write 不覆盖既有键)
166
- try {
167
- const cfgDir = path.join(dshHomeOf(), 'memory', 'semantic')
168
- mkdirSync(cfgDir, { recursive: true })
169
- const cfgPath = path.join(cfgDir, 'embedding-config.json')
170
- let cfg = {}
171
- try { cfg = JSON.parse(await readFile(cfgPath, 'utf8')) } catch (_) {}
172
- if (!cfg || typeof cfg !== 'object') cfg = {}
173
- cfg.gpu = wantGpu
174
- await writeFile(cfgPath, JSON.stringify(cfg, null, 2), 'utf8')
175
- } catch (eCfg) { diagOf('python-setup: gpu pref write failed: ' + String(eCfg && eCfg.message || eCfg)) }
216
+ try { await patchEmbeddingConfig({ gpu: wantGpu }) } catch (eCfg) { diagOf('python-setup: gpu pref write failed: ' + String(eCfg && eCfg.message || eCfg)) }
176
217
  diagOf('python-setup: deps installed into venv (gpu=' + wantGpu + ')')
177
218
  return snapshot()
178
219
  }
@@ -201,6 +242,18 @@ export function createPythonSetupPre(opts = {}) {
201
242
  await downloadWithResume(tu, path.join(modelsDir(), tf), () => {}, () => st.cancelled)
202
243
  }
203
244
  st.modelReady = true; st.phase = 'ready'
245
+ // D1/D2:模型齐备后把**引擎消费所需的键**写全 —— provider(否则 load_embedder 抛 unknown embedding provider)、
246
+ // modelDir(否则 KeyError)、onnxFile(落位是平铺 models/model_int8.onnx,而 worker 默认找 modelDir/onnx/model_int8.onnx)、
247
+ // dimension(向量 identity 块)。tokenizer 五件与 onnx 同在 modelDir 根,AutoTokenizer.from_pretrained(modelDir) 因此可用。
248
+ try {
249
+ await patchEmbeddingConfig({
250
+ provider: PROVIDER_ID_INT8,
251
+ modelDir: modelsDir().replace(/\\/g, '/'),
252
+ onnxFile: path.basename(MODEL_SPEC.file),
253
+ dimension: EMBED_DIMENSION,
254
+ })
255
+ st.configOk = true
256
+ } catch (eCfg) { diagOf('python-setup: embedding-config write failed: ' + String(eCfg && eCfg.message || eCfg)) }
204
257
  diagOf('python-setup: model+tokens ready at ' + modelsDir() + ' (model=' + size + ' bytes, mirror=' + mirror.id + ')')
205
258
  return snapshot()
206
259
  } catch (e) {
@@ -80,7 +80,7 @@ export function pickWindowPre(windows, provider, model) {
80
80
  * @returns {{provider:string,model:string,contextWindow:number}} 未找到时字段为空/0。
81
81
  */
82
82
  export function findSessionModelPre(events, maxScan = 0) {
83
- const out = { provider: '', model: '', contextWindow: 0 }
83
+ const out = { provider: '', model: '', contextWindow: 0, maxTokens: 0 }
84
84
  if (!Array.isArray(events) || !events.length) return out
85
85
  const floor = maxScan > 0 ? Math.max(0, events.length - maxScan) : 0
86
86
  for (let i = events.length - 1; i >= floor; i--) {
@@ -92,13 +92,70 @@ export function findSessionModelPre(events, maxScan = 0) {
92
92
  const provider = String(cfg.provider || d.provider || '')
93
93
  const model = String(cfg.model || d.model || '')
94
94
  const cw = Number(d.contextWindow) || 0
95
+ const mt = Number(cfg.maxTokens) || 0
95
96
  if (!out.model && model) { out.model = model; out.provider = provider }
96
97
  if (!out.contextWindow && cw > 0) out.contextWindow = cw
97
- if (out.model && out.contextWindow) break
98
+ // maxTokens = 路由为**本次请求预留的输出预算**,provider 会把它计入 "requested tokens"(实测:
99
+ // 消息 666,044 + 预留 384,000 = 1,050,044 > 上限 1,048,576 → 400 CONTEXT_WINDOW_EXCEEDED)。
100
+ if (!out.maxTokens && mt > 0) out.maxTokens = mt
101
+ if (out.model && out.contextWindow && out.maxTokens) break
98
102
  }
99
103
  return out
100
104
  }
101
105
 
106
+ /**
107
+ * 水位「硬信号」扫描(2026-09-10,实机取证后新增)。
108
+ *
109
+ * 背景(用户实测:官方压缩又抢在 75% 接续之前):本机 provider 的硬上限是 1,048,576 token,
110
+ * 而**预留输出预算 384,000 也算在请求里** ⇒ 消息实际只能用到约 664,576。插件此前拿
111
+ * `contextWindow`(1,000,000)当分母、拿本地计量当分子,两边都偏乐观,于是 75% 永远到不了,
112
+ * 直到上游直接 400:
113
+ * "This model's maximum context length is 1048576 tokens. However, you requested 1050044 tokens
114
+ * (666044 in the messages, 384000 in the completion)."
115
+ * 这条错误是**唯一权威**的口径来源。本函数把它与 compaction 事件一起扫出来,供上层:
116
+ * ① 解析真实窗口/预留额度/消息实占 → 自我校准分子;
117
+ * ② 一旦发生 overflow 或 compaction,直接当作「已到阈值」触发接续(不再依赖估算)。
118
+ *
119
+ * @param {object[]} events - 会话事件数组。
120
+ * @returns {{reservedTokens:number,overflow:{seq:number,windowTokens:number,requestedTokens:number,messageTokens:number,completionTokens:number,code:string}|null,compactionSeq:number}}
121
+ */
122
+ export function scanPressureSignalsPre(events) {
123
+ const out = { reservedTokens: 0, overflow: null, compactionSeq: 0 }
124
+ if (!Array.isArray(events) || !events.length) return out
125
+ for (let i = events.length - 1; i >= 0; i--) {
126
+ const ev = events[i]
127
+ if (!ev) continue
128
+ const seq = Number(ev.seq) || 0
129
+ const t = ev.type
130
+ if (!out.reservedTokens && (t === 'request/header' || t === 'request/context')) {
131
+ const d = (ev.data || {})
132
+ const cfg = (d.header && d.header.config) || d.config || {}
133
+ const mt = Number(cfg.maxTokens) || 0
134
+ if (mt > 0) out.reservedTokens = mt
135
+ }
136
+ if (!out.compactionSeq && (t === 'compaction/start' || t === 'compaction/summary')) out.compactionSeq = seq
137
+ if (out.overflow && out.compactionSeq && out.reservedTokens) break
138
+ if (out.overflow || t !== 'assistant/attempt') continue
139
+ const stream = (ev.data && ev.data.stream) || []
140
+ for (const item of stream) {
141
+ const chunk = (item && item.chunk) || {}
142
+ if (chunk.type !== 'finish') continue
143
+ const failure = (chunk.reason && chunk.reason.failure) || {}
144
+ const code = String(failure.code || '')
145
+ const msg = String(failure.message || '')
146
+ if (code !== 'CONTEXT_WINDOW_EXCEEDED' && !/maximum context length/i.test(msg)) continue
147
+ const win = Number((msg.match(/maximum context length is\s+(\d+)\s+tokens/i) || [])[1]) || 0
148
+ const req = Number((msg.match(/you requested\s+(\d+)\s+tokens/i) || [])[1]) || 0
149
+ const inMsg = Number((msg.match(/\((\d+)\s+in the messages/i) || [])[1]) || 0
150
+ const inComp = Number((msg.match(/(\d+)\s+in the completion\)/i) || [])[1]) || 0
151
+ out.overflow = { seq, windowTokens: win, requestedTokens: req, messageTokens: inMsg, completionTokens: inComp, code: code || 'CONTEXT_WINDOW_EXCEEDED' }
152
+ break
153
+ }
154
+ }
155
+ return out
156
+ }
157
+
158
+
102
159
  /**
103
160
  * 从会话事件里取最近一条 request/context 的 contextWindow(路由权威值)。
104
161
  * 不限制扫描窗口:该事件只在请求头部变化时追加,长会话里往往只存在于开头。
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@a9i5k4/dsh-auto-memory",
3
3
  "description": "Proactive associative memory for DSH: zero-prompt recall injected before the model speaks, three-layer auto-consolidation, skill crystallization, and Astra-style context management - handoff ledgers, PLAN whiteboard, water-level sensing. Local-first, model-agnostic, zero deps. 主动联想记忆+Astra 式上下文管理:自动唤回/自动沉淀/技能固化/交接账本与白板跨窗口续命/水位感知。",
4
- "version": "2.3.1",
4
+ "version": "2.4.0",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",
7
7
  "exports": {