@a9i5k4/dsh-auto-memory 2.3.1 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +83 -10
- package/lib/context-host.js +22 -4
- package/lib/index.js +630 -327
- package/lib/python-setup.js +67 -14
- package/lib/water-window.js +59 -2
- package/package.json +1 -1
package/lib/python-setup.js
CHANGED
|
@@ -4,9 +4,11 @@
|
|
|
4
4
|
* 四步链路(全部走本模块,UI 只管展示状态与点击):
|
|
5
5
|
* ①detect — 探测系统 Python(≥3.9)/既有 venv/模型本体;全部只读。
|
|
6
6
|
* ②venv — python -m venv <userDir>/python-engine/.venv(幂等:已存在直接跳过)。
|
|
7
|
-
* ③deps — venv 内 pip 安装运行依赖(transformers/onnxruntime
|
|
7
|
+
* ③deps — venv 内 pip 安装运行依赖(transformers/onnxruntime,清华镜像兜底;int8 档不需要 torch)。
|
|
8
8
|
* ④model — BGE-M3 int8(~539MB)下载到 <userDir>/python-engine/models/,
|
|
9
9
|
* cn(hf-mirror)/intl(hf 官方)双通道+SHA256 校验,复用 JS 档下载器的状态机形态。
|
|
10
|
+
* ④' — 模型齐备后**必须回写 embedding-config.json**(provider/modelDir/onnxFile/dimension),
|
|
11
|
+
* 否则 worker load_embedder 抛 unknown embedding provider —— 这是"装了用不了"的头号断点。
|
|
10
12
|
*
|
|
11
13
|
* 设计约束:
|
|
12
14
|
* - 一切落盘在用户目录(~/.dsh/python-engine/),npm 包目录升级会被覆盖,绝不放模型。
|
|
@@ -18,7 +20,7 @@ import { promisify } from 'node:util'
|
|
|
18
20
|
import { createHash } from 'node:crypto'
|
|
19
21
|
import { homedir } from 'node:os'
|
|
20
22
|
import path from 'node:path'
|
|
21
|
-
import { existsSync, mkdirSync, writeFileSync, statSync, createWriteStream } from 'node:fs'
|
|
23
|
+
import { existsSync, mkdirSync, writeFileSync, readFileSync, statSync, createWriteStream } from 'node:fs'
|
|
22
24
|
import { readFile, writeFile, rm } from 'node:fs/promises'
|
|
23
25
|
|
|
24
26
|
const execFileP = promisify(execFile)
|
|
@@ -45,6 +47,15 @@ const TOKENIZER_FILES = ['config.json', 'tokenizer.json', 'tokenizer_config.json
|
|
|
45
47
|
const PIP_DEPS_CPU = ['transformers', 'onnxruntime']
|
|
46
48
|
const PIP_DEPS_GPU_EXTRA = ['onnxruntime-gpu']
|
|
47
49
|
|
|
50
|
+
/** 探针要与安装清单**同口径**:int8 档 load_embedder 只 import numpy+onnxruntime+transformers;
|
|
51
|
+
* 历史上探针多要一个 torch,而 PIP_DEPS_CPU 从不装 torch → depsOk 恒 false,"全部就绪"永不出现。 */
|
|
52
|
+
const DEPS_PROBE = 'import transformers, onnxruntime, numpy; print("deps-ok")'
|
|
53
|
+
|
|
54
|
+
/** worker 侧 load_embedder 的档位名(m7_embedding_v1.PROVIDER_REAL_INT8);发布构建会把 -pre-v1 折成 -v1。 */
|
|
55
|
+
const PROVIDER_ID_INT8 = 'bge-m3-onnx-int8-v1'
|
|
56
|
+
/** dense 维度(BGE-M3 = 1024),写进 embedding-config.json 的 dimension。 */
|
|
57
|
+
const EMBED_DIMENSION = 1024
|
|
58
|
+
|
|
48
59
|
export function createPythonSetupPre(opts = {}) {
|
|
49
60
|
const dshHomeOf = typeof opts.dshHome === 'function' ? opts.dshHome : () => opts.dshHome || path.join(homedir(), '.dsh')
|
|
50
61
|
const diagOf = typeof opts.diag === 'function' ? opts.diag : () => {}
|
|
@@ -64,6 +75,7 @@ export function createPythonSetupPre(opts = {}) {
|
|
|
64
75
|
chosenPython: '',
|
|
65
76
|
venvOk: false,
|
|
66
77
|
depsOk: false,
|
|
78
|
+
configOk: false,
|
|
67
79
|
dl: { bytesDone: 0, bytesTotal: MODEL_SPEC.bytes, mirror: '', startedAt: 0, etaSec: 0 },
|
|
68
80
|
cancelled: false,
|
|
69
81
|
}
|
|
@@ -86,6 +98,43 @@ export function createPythonSetupPre(opts = {}) {
|
|
|
86
98
|
return m ? { major: Number(m[1]), minor: Number(m[2]) } : null
|
|
87
99
|
}
|
|
88
100
|
|
|
101
|
+
// ---------- 引擎侧 embedding-config.json(D1/D2:向导产物必须落在 worker 真正读取的位置与键上) ----------
|
|
102
|
+
/** worker 读 <dsh-home>/memory/semantic/embedding-config.json(见 worker_semantic_v1.load_embedding_config_from_env);
|
|
103
|
+
* 发布构建会把 `semantic` 折成 `semantic`,故这里写带 -pre 的规范名。
|
|
104
|
+
* 历史缺陷:向导曾写 <dsh-home>/memory/semantic/(缺 -pre),pre 线 worker 从不读该路径 → 装完即"找不到配置"。 */
|
|
105
|
+
const embeddingConfigPath = () => path.join(dshHomeOf(), 'memory', 'semantic', 'embedding-config.json')
|
|
106
|
+
|
|
107
|
+
/** read-modify-write:只覆盖本次传入的键,保留引擎运行期自管的 search/activationPolicy/activationEmitMode 等。 */
|
|
108
|
+
async function patchEmbeddingConfig(patch) {
|
|
109
|
+
const cfgPath = embeddingConfigPath()
|
|
110
|
+
mkdirSync(path.dirname(cfgPath), { recursive: true })
|
|
111
|
+
let cfg = {}
|
|
112
|
+
try { cfg = JSON.parse(await readFile(cfgPath, 'utf8')) } catch (_) {}
|
|
113
|
+
if (!cfg || typeof cfg !== 'object' || Array.isArray(cfg)) cfg = {}
|
|
114
|
+
Object.assign(cfg, patch)
|
|
115
|
+
await writeFile(cfgPath, JSON.stringify(cfg, null, 2), 'utf8')
|
|
116
|
+
return cfgPath
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/** 就绪口径(D4):配置声明的 provider/modelDir 必须指向磁盘上真实存在的 onnx + tokenizer;
|
|
120
|
+
* 只看"onnx 已下载"会给出假绿 —— worker 起来照样抛 unknown embedding provider / KeyError modelDir。
|
|
121
|
+
* 注意与下载清单 TOKENIZER_FILES 的区别:那是"下全"的清单(slow/fast 两条路径都覆盖),
|
|
122
|
+
* 这里判的是**能否加载** —— AutoTokenizer 默认 fast 路径有 tokenizer.json 即可,缺 sentencepiece.bpe.model 不算坏。 */
|
|
123
|
+
function configReadyForModels() {
|
|
124
|
+
try {
|
|
125
|
+
const cfg = JSON.parse(readFileSync(embeddingConfigPath(), 'utf8'))
|
|
126
|
+
if (!cfg || typeof cfg !== 'object' || Array.isArray(cfg)) return false
|
|
127
|
+
if (String(cfg.provider || '') !== PROVIDER_ID_INT8) return false
|
|
128
|
+
const base = String(cfg.modelDir || '').trim()
|
|
129
|
+
if (!base) return false
|
|
130
|
+
const rel = String(cfg.onnxFile || 'onnx/model_int8.onnx')
|
|
131
|
+
if (!existsSync(path.join(base, ...rel.split('/')))) return false
|
|
132
|
+
const hasTokenizer = ['tokenizer.json', 'sentencepiece.bpe.model'].some((f) => existsSync(path.join(base, f)))
|
|
133
|
+
if (!hasTokenizer) return false
|
|
134
|
+
return ['config.json', 'tokenizer_config.json'].every((f) => existsSync(path.join(base, f)))
|
|
135
|
+
} catch (_) { return false }
|
|
136
|
+
}
|
|
137
|
+
|
|
89
138
|
/** ①环境探测:venv(推荐)→ 系统 python/python3/py launcher;版本 ≥3.9 且 <3.13(onnxruntime 兼容上界)。 */
|
|
90
139
|
async function detect() {
|
|
91
140
|
st.phase = 'detecting'; st.error = ''
|
|
@@ -106,7 +155,8 @@ export function createPythonSetupPre(opts = {}) {
|
|
|
106
155
|
}
|
|
107
156
|
// 既有成果快照
|
|
108
157
|
st.venvOk = existsSync(venvPython())
|
|
109
|
-
st.depsOk = st.venvOk ? (await probe(venvPython(), ['-c',
|
|
158
|
+
st.depsOk = st.venvOk ? (await probe(venvPython(), ['-c', DEPS_PROBE])).ok : false
|
|
159
|
+
st.configOk = configReadyForModels()
|
|
110
160
|
// modelReady 判定(2026-09-09):onnx + tokenizer 5 件全齐才算就绪,避免 UI 显示 ✓ 但 sidecar 起不来
|
|
111
161
|
st.modelReady = existsSync(modelPath()) && TOKENIZER_FILES.every((f) => existsSync(path.join(modelsDir(), f)))
|
|
112
162
|
st.pythons = out
|
|
@@ -118,7 +168,7 @@ export function createPythonSetupPre(opts = {}) {
|
|
|
118
168
|
return {
|
|
119
169
|
phase: st.phase, error: st.error,
|
|
120
170
|
pythons: st.pythons, chosenPython: st.chosenPython,
|
|
121
|
-
venvOk: st.venvOk, depsOk: st.depsOk, modelReady: st.modelReady, wantGpu: !!st.wantGpu,
|
|
171
|
+
venvOk: st.venvOk, depsOk: st.depsOk, configOk: st.configOk, modelReady: st.modelReady, wantGpu: !!st.wantGpu,
|
|
122
172
|
modelPath: modelPath(), venvPython: venvPython(),
|
|
123
173
|
modelBytes: st.modelReady ? statSync(modelPath()).size : 0, modelExpectedBytes: MODEL_SPEC.bytes,
|
|
124
174
|
dl: { ...st.dl },
|
|
@@ -163,16 +213,7 @@ export function createPythonSetupPre(opts = {}) {
|
|
|
163
213
|
st.depsOk = !!r.ok
|
|
164
214
|
if (!r.ok) { st.error = '依赖安装失败: ' + (r.tail || ''); st.phase = 'error'; return snapshot() }
|
|
165
215
|
// GPU 偏好写入 embedding-config.json(worker 读 config['gpu'] 选 CUDA/CPU provider;read-modify-write 不覆盖既有键)
|
|
166
|
-
try {
|
|
167
|
-
const cfgDir = path.join(dshHomeOf(), 'memory', 'semantic')
|
|
168
|
-
mkdirSync(cfgDir, { recursive: true })
|
|
169
|
-
const cfgPath = path.join(cfgDir, 'embedding-config.json')
|
|
170
|
-
let cfg = {}
|
|
171
|
-
try { cfg = JSON.parse(await readFile(cfgPath, 'utf8')) } catch (_) {}
|
|
172
|
-
if (!cfg || typeof cfg !== 'object') cfg = {}
|
|
173
|
-
cfg.gpu = wantGpu
|
|
174
|
-
await writeFile(cfgPath, JSON.stringify(cfg, null, 2), 'utf8')
|
|
175
|
-
} catch (eCfg) { diagOf('python-setup: gpu pref write failed: ' + String(eCfg && eCfg.message || eCfg)) }
|
|
216
|
+
try { await patchEmbeddingConfig({ gpu: wantGpu }) } catch (eCfg) { diagOf('python-setup: gpu pref write failed: ' + String(eCfg && eCfg.message || eCfg)) }
|
|
176
217
|
diagOf('python-setup: deps installed into venv (gpu=' + wantGpu + ')')
|
|
177
218
|
return snapshot()
|
|
178
219
|
}
|
|
@@ -201,6 +242,18 @@ export function createPythonSetupPre(opts = {}) {
|
|
|
201
242
|
await downloadWithResume(tu, path.join(modelsDir(), tf), () => {}, () => st.cancelled)
|
|
202
243
|
}
|
|
203
244
|
st.modelReady = true; st.phase = 'ready'
|
|
245
|
+
// D1/D2:模型齐备后把**引擎消费所需的键**写全 —— provider(否则 load_embedder 抛 unknown embedding provider)、
|
|
246
|
+
// modelDir(否则 KeyError)、onnxFile(落位是平铺 models/model_int8.onnx,而 worker 默认找 modelDir/onnx/model_int8.onnx)、
|
|
247
|
+
// dimension(向量 identity 块)。tokenizer 五件与 onnx 同在 modelDir 根,AutoTokenizer.from_pretrained(modelDir) 因此可用。
|
|
248
|
+
try {
|
|
249
|
+
await patchEmbeddingConfig({
|
|
250
|
+
provider: PROVIDER_ID_INT8,
|
|
251
|
+
modelDir: modelsDir().replace(/\\/g, '/'),
|
|
252
|
+
onnxFile: path.basename(MODEL_SPEC.file),
|
|
253
|
+
dimension: EMBED_DIMENSION,
|
|
254
|
+
})
|
|
255
|
+
st.configOk = true
|
|
256
|
+
} catch (eCfg) { diagOf('python-setup: embedding-config write failed: ' + String(eCfg && eCfg.message || eCfg)) }
|
|
204
257
|
diagOf('python-setup: model+tokens ready at ' + modelsDir() + ' (model=' + size + ' bytes, mirror=' + mirror.id + ')')
|
|
205
258
|
return snapshot()
|
|
206
259
|
} catch (e) {
|
package/lib/water-window.js
CHANGED
|
@@ -80,7 +80,7 @@ export function pickWindowPre(windows, provider, model) {
|
|
|
80
80
|
* @returns {{provider:string,model:string,contextWindow:number}} 未找到时字段为空/0。
|
|
81
81
|
*/
|
|
82
82
|
export function findSessionModelPre(events, maxScan = 0) {
|
|
83
|
-
const out = { provider: '', model: '', contextWindow: 0 }
|
|
83
|
+
const out = { provider: '', model: '', contextWindow: 0, maxTokens: 0 }
|
|
84
84
|
if (!Array.isArray(events) || !events.length) return out
|
|
85
85
|
const floor = maxScan > 0 ? Math.max(0, events.length - maxScan) : 0
|
|
86
86
|
for (let i = events.length - 1; i >= floor; i--) {
|
|
@@ -92,13 +92,70 @@ export function findSessionModelPre(events, maxScan = 0) {
|
|
|
92
92
|
const provider = String(cfg.provider || d.provider || '')
|
|
93
93
|
const model = String(cfg.model || d.model || '')
|
|
94
94
|
const cw = Number(d.contextWindow) || 0
|
|
95
|
+
const mt = Number(cfg.maxTokens) || 0
|
|
95
96
|
if (!out.model && model) { out.model = model; out.provider = provider }
|
|
96
97
|
if (!out.contextWindow && cw > 0) out.contextWindow = cw
|
|
97
|
-
|
|
98
|
+
// maxTokens = 路由为**本次请求预留的输出预算**,provider 会把它计入 "requested tokens"(实测:
|
|
99
|
+
// 消息 666,044 + 预留 384,000 = 1,050,044 > 上限 1,048,576 → 400 CONTEXT_WINDOW_EXCEEDED)。
|
|
100
|
+
if (!out.maxTokens && mt > 0) out.maxTokens = mt
|
|
101
|
+
if (out.model && out.contextWindow && out.maxTokens) break
|
|
98
102
|
}
|
|
99
103
|
return out
|
|
100
104
|
}
|
|
101
105
|
|
|
106
|
+
/**
|
|
107
|
+
* 水位「硬信号」扫描(2026-09-10,实机取证后新增)。
|
|
108
|
+
*
|
|
109
|
+
* 背景(用户实测:官方压缩又抢在 75% 接续之前):本机 provider 的硬上限是 1,048,576 token,
|
|
110
|
+
* 而**预留输出预算 384,000 也算在请求里** ⇒ 消息实际只能用到约 664,576。插件此前拿
|
|
111
|
+
* `contextWindow`(1,000,000)当分母、拿本地计量当分子,两边都偏乐观,于是 75% 永远到不了,
|
|
112
|
+
* 直到上游直接 400:
|
|
113
|
+
* "This model's maximum context length is 1048576 tokens. However, you requested 1050044 tokens
|
|
114
|
+
* (666044 in the messages, 384000 in the completion)."
|
|
115
|
+
* 这条错误是**唯一权威**的口径来源。本函数把它与 compaction 事件一起扫出来,供上层:
|
|
116
|
+
* ① 解析真实窗口/预留额度/消息实占 → 自我校准分子;
|
|
117
|
+
* ② 一旦发生 overflow 或 compaction,直接当作「已到阈值」触发接续(不再依赖估算)。
|
|
118
|
+
*
|
|
119
|
+
* @param {object[]} events - 会话事件数组。
|
|
120
|
+
* @returns {{reservedTokens:number,overflow:{seq:number,windowTokens:number,requestedTokens:number,messageTokens:number,completionTokens:number,code:string}|null,compactionSeq:number}}
|
|
121
|
+
*/
|
|
122
|
+
export function scanPressureSignalsPre(events) {
|
|
123
|
+
const out = { reservedTokens: 0, overflow: null, compactionSeq: 0 }
|
|
124
|
+
if (!Array.isArray(events) || !events.length) return out
|
|
125
|
+
for (let i = events.length - 1; i >= 0; i--) {
|
|
126
|
+
const ev = events[i]
|
|
127
|
+
if (!ev) continue
|
|
128
|
+
const seq = Number(ev.seq) || 0
|
|
129
|
+
const t = ev.type
|
|
130
|
+
if (!out.reservedTokens && (t === 'request/header' || t === 'request/context')) {
|
|
131
|
+
const d = (ev.data || {})
|
|
132
|
+
const cfg = (d.header && d.header.config) || d.config || {}
|
|
133
|
+
const mt = Number(cfg.maxTokens) || 0
|
|
134
|
+
if (mt > 0) out.reservedTokens = mt
|
|
135
|
+
}
|
|
136
|
+
if (!out.compactionSeq && (t === 'compaction/start' || t === 'compaction/summary')) out.compactionSeq = seq
|
|
137
|
+
if (out.overflow && out.compactionSeq && out.reservedTokens) break
|
|
138
|
+
if (out.overflow || t !== 'assistant/attempt') continue
|
|
139
|
+
const stream = (ev.data && ev.data.stream) || []
|
|
140
|
+
for (const item of stream) {
|
|
141
|
+
const chunk = (item && item.chunk) || {}
|
|
142
|
+
if (chunk.type !== 'finish') continue
|
|
143
|
+
const failure = (chunk.reason && chunk.reason.failure) || {}
|
|
144
|
+
const code = String(failure.code || '')
|
|
145
|
+
const msg = String(failure.message || '')
|
|
146
|
+
if (code !== 'CONTEXT_WINDOW_EXCEEDED' && !/maximum context length/i.test(msg)) continue
|
|
147
|
+
const win = Number((msg.match(/maximum context length is\s+(\d+)\s+tokens/i) || [])[1]) || 0
|
|
148
|
+
const req = Number((msg.match(/you requested\s+(\d+)\s+tokens/i) || [])[1]) || 0
|
|
149
|
+
const inMsg = Number((msg.match(/\((\d+)\s+in the messages/i) || [])[1]) || 0
|
|
150
|
+
const inComp = Number((msg.match(/(\d+)\s+in the completion\)/i) || [])[1]) || 0
|
|
151
|
+
out.overflow = { seq, windowTokens: win, requestedTokens: req, messageTokens: inMsg, completionTokens: inComp, code: code || 'CONTEXT_WINDOW_EXCEEDED' }
|
|
152
|
+
break
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
return out
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
|
|
102
159
|
/**
|
|
103
160
|
* 从会话事件里取最近一条 request/context 的 contextWindow(路由权威值)。
|
|
104
161
|
* 不限制扫描窗口:该事件只在请求头部变化时追加,长会话里往往只存在于开头。
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@a9i5k4/dsh-auto-memory",
|
|
3
3
|
"description": "Proactive associative memory for DSH: zero-prompt recall injected before the model speaks, three-layer auto-consolidation, skill crystallization, and Astra-style context management - handoff ledgers, PLAN whiteboard, water-level sensing. Local-first, model-agnostic, zero deps. 主动联想记忆+Astra 式上下文管理:自动唤回/自动沉淀/技能固化/交接账本与白板跨窗口续命/水位感知。",
|
|
4
|
-
"version": "2.
|
|
4
|
+
"version": "2.4.0",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
7
7
|
"exports": {
|