mingdao-harness 0.1.54
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +246 -0
- package/assets/tokenizer-data.json.gz +0 -0
- package/docs/ARCHITECTURE.md +108 -0
- package/docs/CONFIG.md +257 -0
- package/docs/DESKTOP-EVALUATION.md +48 -0
- package/docs/PROVIDERS.md +98 -0
- package/docs/QA-REPORT.md +333 -0
- package/install.bat +8 -0
- package/install.ps1 +57 -0
- package/install.sh +154 -0
- package/package.json +61 -0
- package/skills/api-design/SKILL.md +21 -0
- package/skills/code-review/SKILL.md +31 -0
- package/skills/debugging/SKILL.md +21 -0
- package/skills/docker/SKILL.md +27 -0
- package/skills/docx/SKILL.md +28 -0
- package/skills/frontend-design/SKILL.md +31 -0
- package/skills/git-commit/SKILL.md +32 -0
- package/skills/pdf/SKILL.md +30 -0
- package/skills/pptx/SKILL.md +24 -0
- package/skills/refactoring/SKILL.md +24 -0
- package/skills/release-checklist/SKILL.md +20 -0
- package/skills/testing/SKILL.md +30 -0
- package/skills/webapp-testing/SKILL.md +27 -0
- package/skills/xlsx/SKILL.md +28 -0
- package/src/agent.js +472 -0
- package/src/audit.js +68 -0
- package/src/autostart.js +74 -0
- package/src/batch.js +182 -0
- package/src/cachestats.js +215 -0
- package/src/cli.js +1099 -0
- package/src/commands/key.js +75 -0
- package/src/commands/schedule.js +176 -0
- package/src/commands/skill.js +168 -0
- package/src/commands/sync.js +222 -0
- package/src/commands/update.js +157 -0
- package/src/commands/workspace.js +109 -0
- package/src/compact.js +112 -0
- package/src/config.js +134 -0
- package/src/context.js +86 -0
- package/src/cost-guard.js +58 -0
- package/src/credentials.js +69 -0
- package/src/hooks.js +123 -0
- package/src/index.js +42 -0
- package/src/mcp-presets.js +78 -0
- package/src/mcp.js +284 -0
- package/src/memory.js +242 -0
- package/src/model-discovery.js +153 -0
- package/src/models.js +186 -0
- package/src/notify.js +38 -0
- package/src/permissions.js +83 -0
- package/src/pricing.js +156 -0
- package/src/prompts.js +58 -0
- package/src/providers/index.js +135 -0
- package/src/providers/openai-compatible.js +171 -0
- package/src/routing.js +126 -0
- package/src/schedule.js +434 -0
- package/src/session-index.js +122 -0
- package/src/session.js +131 -0
- package/src/skill-lib.js +350 -0
- package/src/skill-registry.js +165 -0
- package/src/skills.js +135 -0
- package/src/sync-server.js +564 -0
- package/src/sync.js +479 -0
- package/src/tasks.js +126 -0
- package/src/titles.js +75 -0
- package/src/tokenizer.js +287 -0
- package/src/tools/bash.js +165 -0
- package/src/tools/fs-tools.js +346 -0
- package/src/tools/index.js +287 -0
- package/src/ui.js +643 -0
- package/src/update.js +222 -0
- package/src/web/attachments.js +49 -0
- package/src/web/index.html +993 -0
- package/src/web/server.js +1173 -0
- package/src/web/web-io.js +107 -0
- package/src/workspace.js +157 -0
package/src/tokenizer.js
ADDED
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
// 精确 tokenizer(零运行时依赖):
|
|
2
|
+
// - 内置 DeepSeek 官方词表(assets/tokenizer-data.json.gz,源自 DeepSeek-V3 tokenizer.json)
|
|
3
|
+
// - 字节级 BPE 计数:added_tokens 合并正则一次扫描(O(n),替代逐 token indexOf)+ 官方 Split 预分词 + 按 rank 合并
|
|
4
|
+
// - 非 DeepSeek 模型回退启发式估算(英文≈4字符/token,CJK≈0.75 token/字,其余非 ASCII≈1)
|
|
5
|
+
// - 内容级计数缓存:同一文本(如多轮不变的会话消息)只做一次 BPE,重复调用 O(1) 命中
|
|
6
|
+
// 仅用于上下文预算计数,不输出 token id。
|
|
7
|
+
|
|
8
|
+
import fs from 'node:fs';
|
|
9
|
+
import path from 'node:path';
|
|
10
|
+
import zlib from 'node:zlib';
|
|
11
|
+
import { fileURLToPath } from 'node:url';
|
|
12
|
+
import { mingdaoHome } from './config.js';
|
|
13
|
+
|
|
14
|
+
// DeepSeek 官方预分词(tokenizer.json 的 Split 序列,与 HF tokenizers 语义一致):
|
|
15
|
+
// 1. \p{N}{1,3} 数字按 1–3 位切成独立段
|
|
16
|
+
// 2. [一-龥-ゟ゠-ヿ]+ 中日韩表意文字/假名连续段
|
|
17
|
+
// 3. 标点引导的词|字母段(可带一个前导非字母)|标点串|换行|空白
|
|
18
|
+
// 每级 Split(Isolated) 对上一级全部片段再切分,匹配段与间隔段都保留为独立预分词。
|
|
19
|
+
const SPLIT_RES = [
|
|
20
|
+
/\p{N}{1,3}/gu,
|
|
21
|
+
/[一-龥-ゟ゠-ヿ]+/gu,
|
|
22
|
+
/[!"#$%&'()*+,\-./:;<=>?@\[\\\]^_`{|}~][A-Za-z]+|[^\r\n\p{L}\p{P}\p{S}]?[\p{L}\p{M}]+| ?[\p{P}\p{S}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+/gu,
|
|
23
|
+
];
|
|
24
|
+
|
|
25
|
+
function pretokenize(text) {
|
|
26
|
+
let pieces = [text];
|
|
27
|
+
for (const re of SPLIT_RES) {
|
|
28
|
+
const next = [];
|
|
29
|
+
for (const p of pieces) {
|
|
30
|
+
let last = 0;
|
|
31
|
+
for (const m of p.matchAll(re)) {
|
|
32
|
+
if (m.index > last) next.push(p.slice(last, m.index));
|
|
33
|
+
if (m[0]) next.push(m[0]);
|
|
34
|
+
last = m.index + m[0].length;
|
|
35
|
+
}
|
|
36
|
+
if (last < p.length) next.push(p.slice(last));
|
|
37
|
+
}
|
|
38
|
+
pieces = next;
|
|
39
|
+
}
|
|
40
|
+
return pieces;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// GPT-2 字节到 Unicode 的映射表(byte_to_unicode)。
|
|
44
|
+
// HF tokenizer.json 的 merges/vocab 使用映射后的可打印字符表示:
|
|
45
|
+
// 可打印区间(0x21-0x7E、0xA1-0xAC、0xAE-0xFF)映射为自身,
|
|
46
|
+
// 其余字节(控制符、空格、0x7F-0xA0、0xAD)依次映射到 U+0100+n。
|
|
47
|
+
// 运行时符号必须与词表同表示,否则 73% 的 merge 对(含映射字符)永远匹配不到,
|
|
48
|
+
// 汉字会退化为逐字节计数(如「的」被计为 3 tokens 而非 1)。
|
|
49
|
+
const BYTE_TO_UNICODE = (() => {
|
|
50
|
+
const m = new Array(256);
|
|
51
|
+
const self = (b) => (b >= 0x21 && b <= 0x7e) || (b >= 0xa1 && b <= 0xac) || (b >= 0xae);
|
|
52
|
+
let n = 0;
|
|
53
|
+
for (let b = 0; b < 256; b++) {
|
|
54
|
+
if (self(b)) m[b] = String.fromCharCode(b);
|
|
55
|
+
else {
|
|
56
|
+
m[b] = String.fromCharCode(256 + n);
|
|
57
|
+
n += 1;
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
return m;
|
|
61
|
+
})();
|
|
62
|
+
|
|
63
|
+
function escapeRe(s) {
|
|
64
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
let data = null;
|
|
68
|
+
let loadError = null;
|
|
69
|
+
|
|
70
|
+
function loadData() {
|
|
71
|
+
if (data || loadError) return data;
|
|
72
|
+
try {
|
|
73
|
+
const file = fileURLToPath(new URL('../assets/tokenizer-data.json.gz', import.meta.url));
|
|
74
|
+
const raw = zlib.gunzipSync(fs.readFileSync(file)).toString('utf8');
|
|
75
|
+
const parsed = JSON.parse(raw);
|
|
76
|
+
const mergeRank = new Map();
|
|
77
|
+
for (let i = 0; i < parsed.merges.length; i++) {
|
|
78
|
+
const [a, b] = parsed.merges[i];
|
|
79
|
+
mergeRank.set(a + '\u0001' + b, i);
|
|
80
|
+
}
|
|
81
|
+
const added = (parsed.added || []).filter(Boolean).sort((a, b) => b.length - a.length);
|
|
82
|
+
// 全部 added token 合并成单个正则(按长度降序排列 → 左起最长优先,与逐 token startsWith 语义一致),
|
|
83
|
+
// 一次 matchAll 定位所有特殊 token,把 O(文本长 × 818 个 indexOf) 降到 O(n)
|
|
84
|
+
const addedRe = added.length ? new RegExp(added.map(escapeRe).join('|'), 'gu') : null;
|
|
85
|
+
data = { mergeRank, added, addedRe };
|
|
86
|
+
} catch (err) {
|
|
87
|
+
loadError = err;
|
|
88
|
+
}
|
|
89
|
+
return data;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export function isTokenizable(modelName) {
|
|
93
|
+
if (typeof modelName !== 'string') return false;
|
|
94
|
+
if (modelName.startsWith('deepseek')) return true;
|
|
95
|
+
// B8(评估建议):自定义端点跑 DeepSeek 系模型时,config.customModels.<name>.tokenizer = "deepseek"
|
|
96
|
+
// 即可按官方词表精确计数(否则回退启发式,预算误差可达 ±2 倍)
|
|
97
|
+
return customTokenizerNames().has(modelName);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// 配置按 mtime 缓存:避免每次计数都读盘解析 config.json
|
|
101
|
+
let customTokCache = { mtime: 0, names: new Set() };
|
|
102
|
+
function customTokenizerNames() {
|
|
103
|
+
try {
|
|
104
|
+
const file = path.join(mingdaoHome(), 'config.json');
|
|
105
|
+
const st = fs.statSync(file);
|
|
106
|
+
if (st.mtimeMs === customTokCache.mtime) return customTokCache.names;
|
|
107
|
+
const cfg = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
108
|
+
const names = new Set();
|
|
109
|
+
for (const [n, c] of Object.entries(cfg?.customModels || {})) {
|
|
110
|
+
if (c && c.tokenizer === 'deepseek') names.add(n);
|
|
111
|
+
}
|
|
112
|
+
customTokCache = { mtime: st.mtimeMs, names };
|
|
113
|
+
} catch {
|
|
114
|
+
// 无配置/解析失败 → 保持空集合(下次重试)
|
|
115
|
+
}
|
|
116
|
+
return customTokCache.names;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
// 启发式估算(无词表模型的回退路径)。
|
|
120
|
+
// CJK 校准:主流模型流畅中文实测 ≈0.5–0.75 token/字(词表含多字词),
|
|
121
|
+
// 旧版「1 字 = 1 token」会高估约 2 倍、过早触发预算裁剪;取 0.75 保守上界。
|
|
122
|
+
// 其余非 ASCII(emoji/符号)保持 1(多数词表下单个 emoji 常为 2–3 token,不低估)。
|
|
123
|
+
const CJK_RANGES = [
|
|
124
|
+
[0x3400, 0x4dbf], [0x4e00, 0x9fff], [0xf900, 0xfaff], // CJK 扩展/基本区/兼容
|
|
125
|
+
[0x3040, 0x30ff], [0xac00, 0xd7af], // 假名 / 谚文
|
|
126
|
+
];
|
|
127
|
+
const isCjk = (code) => CJK_RANGES.some(([lo, hi]) => code >= lo && code <= hi);
|
|
128
|
+
|
|
129
|
+
export function heuristicTokens(text) {
|
|
130
|
+
if (!text) return 0;
|
|
131
|
+
let ascii = 0;
|
|
132
|
+
let cjk = 0;
|
|
133
|
+
let other = 0;
|
|
134
|
+
for (const ch of String(text)) {
|
|
135
|
+
const code = ch.codePointAt(0);
|
|
136
|
+
if (code < 128) ascii += 1;
|
|
137
|
+
else if (isCjk(code)) cjk += 1;
|
|
138
|
+
else other += code > 0xffff ? 2 : 1; // 审计 B5:增补平面 emoji 按 2 token 保守计
|
|
139
|
+
}
|
|
140
|
+
// 审计 B5:非 CJK 非 ASCII(emoji 等)按码点计但每个 2 个 UTF-16 单元的 emoji 计 2,
|
|
141
|
+
// 避免对预算的过度乐观(ZWJ 序列仍可能低估,但方向已保守)
|
|
142
|
+
return Math.ceil(ascii / 4 + cjk * 0.75 + other);
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
// 单个预分词片段的 BPE 计数(tiktoken 语义:优先合并 rank 最小的对,同 rank 取最左)。
|
|
146
|
+
// 惰性最小堆实现(O(n log n)):大文本(中文长文/工具输出)不再 O(n²) 全表扫描。
|
|
147
|
+
function countPiece(piece, d) {
|
|
148
|
+
const bytes = Buffer.from(piece, 'utf8');
|
|
149
|
+
const syms = [];
|
|
150
|
+
for (const b of bytes) syms.push(BYTE_TO_UNICODE[b]); // 与词表同表示(GPT-2 字节映射)
|
|
151
|
+
if (syms.length <= 1) return syms.length;
|
|
152
|
+
|
|
153
|
+
const alive = new Uint8Array(syms.length).fill(1);
|
|
154
|
+
const leftOf = (i) => {
|
|
155
|
+
for (let j = i - 1; j >= 0; j--) if (alive[j]) return j;
|
|
156
|
+
return -1;
|
|
157
|
+
};
|
|
158
|
+
const rightOf = (i) => {
|
|
159
|
+
for (let j = i + 1; j < syms.length; j++) if (alive[j]) return j;
|
|
160
|
+
return -1;
|
|
161
|
+
};
|
|
162
|
+
|
|
163
|
+
// 最小堆(rank, leftIndex);过期条目惰性丢弃/重推
|
|
164
|
+
const heap = [];
|
|
165
|
+
const push = (rank, idx) => {
|
|
166
|
+
heap.push([rank, idx]);
|
|
167
|
+
let c = heap.length - 1;
|
|
168
|
+
while (c > 0) {
|
|
169
|
+
const p = (c - 1) >> 1;
|
|
170
|
+
if (heap[p][0] < heap[c][0] || (heap[p][0] === heap[c][0] && heap[p][1] <= heap[c][1])) break;
|
|
171
|
+
[heap[p], heap[c]] = [heap[c], heap[p]];
|
|
172
|
+
c = p;
|
|
173
|
+
}
|
|
174
|
+
};
|
|
175
|
+
const pop = () => {
|
|
176
|
+
const top = heap[0];
|
|
177
|
+
const last = heap.pop();
|
|
178
|
+
if (heap.length) {
|
|
179
|
+
heap[0] = last;
|
|
180
|
+
let c = 0;
|
|
181
|
+
for (;;) {
|
|
182
|
+
const l = c * 2 + 1;
|
|
183
|
+
const r = l + 1;
|
|
184
|
+
let m = c;
|
|
185
|
+
const better = (x) => heap[x][0] < heap[m][0] || (heap[x][0] === heap[m][0] && heap[x][1] < heap[m][1]);
|
|
186
|
+
if (l < heap.length && better(l)) m = l;
|
|
187
|
+
if (r < heap.length && better(r)) m = r;
|
|
188
|
+
if (m === c) break;
|
|
189
|
+
[heap[m], heap[c]] = [heap[c], heap[m]];
|
|
190
|
+
c = m;
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
return top;
|
|
194
|
+
};
|
|
195
|
+
|
|
196
|
+
for (let i = 0; i < syms.length - 1; i++) {
|
|
197
|
+
const r = d.mergeRank.get(syms[i] + '\u0001' + syms[i + 1]);
|
|
198
|
+
if (r !== undefined) push(r, i);
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
for (;;) {
|
|
202
|
+
let pair = null;
|
|
203
|
+
while (heap.length) {
|
|
204
|
+
const [rank, idx] = pop();
|
|
205
|
+
if (!alive[idx]) continue;
|
|
206
|
+
const right = rightOf(idx);
|
|
207
|
+
if (right === -1) continue;
|
|
208
|
+
const r = d.mergeRank.get(syms[idx] + '\u0001' + syms[right]);
|
|
209
|
+
if (r === undefined) continue;
|
|
210
|
+
if (r === rank) {
|
|
211
|
+
pair = [idx, right];
|
|
212
|
+
break;
|
|
213
|
+
}
|
|
214
|
+
push(r, idx); // rank 过期(邻居变化):重推
|
|
215
|
+
}
|
|
216
|
+
if (!pair) break;
|
|
217
|
+
const [li, ri] = pair;
|
|
218
|
+
syms[li] = syms[li] + syms[ri];
|
|
219
|
+
alive[ri] = 0;
|
|
220
|
+
const left = leftOf(li);
|
|
221
|
+
if (left !== -1) {
|
|
222
|
+
const r = d.mergeRank.get(syms[left] + '\u0001' + syms[li]);
|
|
223
|
+
if (r !== undefined) push(r, left);
|
|
224
|
+
}
|
|
225
|
+
const right = rightOf(li);
|
|
226
|
+
if (right !== -1) {
|
|
227
|
+
const r = d.mergeRank.get(syms[li] + '\u0001' + syms[right]);
|
|
228
|
+
if (r !== undefined) push(r, li);
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
return alive.reduce((s, v) => s + v, 0);
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
// 内容级计数缓存(modelName → 文本 → token 数):多轮会话中历史消息内容不变,
|
|
235
|
+
// 每步 trimMessages 重复计数同一文本时直接命中。上限保护:超长文本不进缓存、
|
|
236
|
+
// 每模型 512 条封顶(溢出整体清空,简单 LRU 退化策略)。
|
|
237
|
+
const TOKEN_CACHE_MAX_ENTRIES = 512;
|
|
238
|
+
const TOKEN_CACHE_MAX_LEN = 50000;
|
|
239
|
+
const tokenCache = new Map();
|
|
240
|
+
|
|
241
|
+
function countDeepseek(s) {
|
|
242
|
+
const d = loadData();
|
|
243
|
+
if (!d) return heuristicTokens(s); // 词表缺失时优雅回退
|
|
244
|
+
let total = 0;
|
|
245
|
+
let pos = 0;
|
|
246
|
+
if (d.addedRe) {
|
|
247
|
+
for (const m of s.matchAll(d.addedRe)) {
|
|
248
|
+
if (m.index > pos) total += countGap(s.slice(pos, m.index), d);
|
|
249
|
+
total += 1; // added token 自身计 1
|
|
250
|
+
pos = m.index + m[0].length;
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
if (pos < s.length) total += countGap(s.slice(pos), d);
|
|
254
|
+
return total;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
function countGap(piece, d) {
|
|
258
|
+
let total = 0;
|
|
259
|
+
for (const m of pretokenize(piece)) total += countPiece(m, d);
|
|
260
|
+
return total;
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
export function countTokens(text, modelName) {
|
|
264
|
+
if (!text) return 0;
|
|
265
|
+
const s = String(text);
|
|
266
|
+
if (!isTokenizable(modelName)) return heuristicTokens(s);
|
|
267
|
+
if (s.length > TOKEN_CACHE_MAX_LEN) return countDeepseek(s);
|
|
268
|
+
let byModel = tokenCache.get(modelName);
|
|
269
|
+
if (!byModel) {
|
|
270
|
+
byModel = new Map();
|
|
271
|
+
tokenCache.set(modelName, byModel);
|
|
272
|
+
}
|
|
273
|
+
const hit = byModel.get(s);
|
|
274
|
+
if (hit !== undefined) return hit;
|
|
275
|
+
const n = countDeepseek(s);
|
|
276
|
+
if (byModel.size >= TOKEN_CACHE_MAX_ENTRIES) byModel.clear();
|
|
277
|
+
byModel.set(s, n);
|
|
278
|
+
return n;
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
// 供上下文预算使用的计数器工厂
|
|
282
|
+
export function makeTokenCounter(modelName) {
|
|
283
|
+
if (isTokenizable(modelName)) {
|
|
284
|
+
return (text) => countTokens(text, modelName);
|
|
285
|
+
}
|
|
286
|
+
return (text) => heuristicTokens(text);
|
|
287
|
+
}
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
// bash 工具:在子进程中执行 shell 命令,超时强杀,输出截断。
|
|
2
|
+
// 沙箱模式(Linux + bubblewrap):
|
|
3
|
+
// off 直接执行(默认,兼容原有行为)
|
|
4
|
+
// readonly 全盘只读 + /tmp 可写(tmpfs),网络可用
|
|
5
|
+
// safe 只读文件系统 + 断网(unshare-net),工作目录可写 + /tmp 可写
|
|
6
|
+
// 非 Linux 或未安装 bwrap 时自动降级为 off,并在结果中注明(不静默假装沙箱)。
|
|
7
|
+
|
|
8
|
+
import { spawn, spawnSync } from 'node:child_process';
|
|
9
|
+
|
|
10
|
+
const MAX_OUTPUT = 20000;
|
|
11
|
+
const MAX_TIMEOUT_SECONDS = 600;
|
|
12
|
+
|
|
13
|
+
// 敏感环境变量过滤(P1-5 + 评估 P2-3):默认常开——模型驱动的命令不应直接读到 API Key/凭证
|
|
14
|
+
// (一条 env 即可泄露),与沙箱档位解耦;config.bashEnvKeep 按名放行,config.bashEnvFilter=false
|
|
15
|
+
// 整体关闭(回到完全透传)。
|
|
16
|
+
const SENSITIVE_ENV_PAIR = /(api[_-]?key|access[_-]?key|client[_-]?secret|private[_-]?key)/i;
|
|
17
|
+
const SENSITIVE_ENV_SEGMENT = /(^|_)(token|secret|password|passwd|credential|authorization|auth)(_|$)/i;
|
|
18
|
+
const isSensitiveEnv = (k) => SENSITIVE_ENV_PAIR.test(k) || SENSITIVE_ENV_SEGMENT.test(k);
|
|
19
|
+
|
|
20
|
+
function buildChildEnv(ctx, filterSensitive) {
|
|
21
|
+
if (!filterSensitive) return process.env;
|
|
22
|
+
const keep = new Set((ctx?.cfg?.bashEnvKeep || []).map(String));
|
|
23
|
+
const env = {};
|
|
24
|
+
for (const [k, v] of Object.entries(process.env)) {
|
|
25
|
+
if (!isSensitiveEnv(k) || keep.has(k)) env[k] = v;
|
|
26
|
+
}
|
|
27
|
+
return env;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
let sandboxSupport = null;
|
|
31
|
+
|
|
32
|
+
export function detectSandbox() {
|
|
33
|
+
if (sandboxSupport !== null) return sandboxSupport;
|
|
34
|
+
sandboxSupport = 'none';
|
|
35
|
+
if (process.platform === 'linux') {
|
|
36
|
+
try {
|
|
37
|
+
const r = spawnSync('bwrap', ['--version'], { stdio: 'ignore', timeout: 3000 });
|
|
38
|
+
sandboxSupport = r.error ? 'none' : 'bwrap';
|
|
39
|
+
} catch {
|
|
40
|
+
sandboxSupport = 'none';
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
return sandboxSupport;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function tail(s, n) {
|
|
47
|
+
return s.length > n ? `…[输出过长,已截断头部]\n${s.slice(-n)}` : s;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
export function runBash(args, ctx) {
|
|
51
|
+
const command = String(args.command ?? '');
|
|
52
|
+
if (!command.trim()) return { ok: false, error: 'command 参数为空。' };
|
|
53
|
+
const timeoutSec = Math.min(Number(args.timeout) || 120, MAX_TIMEOUT_SECONDS);
|
|
54
|
+
// 配置优先:模型不能通过传 sandbox:'off' 自行降级(配置里选了 safe/readonly 就必须沙箱)
|
|
55
|
+
const mode = String(ctx?.cfg?.sandbox ?? args.sandbox ?? 'off');
|
|
56
|
+
const shell = process.platform === 'win32' ? 'cmd.exe' : '/bin/bash';
|
|
57
|
+
const shellArgs = process.platform === 'win32' ? ['/d', '/s', '/c', command] : ['-lc', command];
|
|
58
|
+
|
|
59
|
+
let spawnCmd = shell;
|
|
60
|
+
let spawnArgs = shellArgs;
|
|
61
|
+
let sandbox = 'off';
|
|
62
|
+
let note = '';
|
|
63
|
+
|
|
64
|
+
if (mode !== 'off' && process.platform === 'linux' && detectSandbox() === 'bwrap') {
|
|
65
|
+
const base = [
|
|
66
|
+
'--die-with-parent',
|
|
67
|
+
'--new-session',
|
|
68
|
+
'--ro-bind', '/', '/',
|
|
69
|
+
'--dev', '/dev',
|
|
70
|
+
'--proc', '/proc',
|
|
71
|
+
'--tmpfs', '/tmp',
|
|
72
|
+
];
|
|
73
|
+
if (mode === 'safe') {
|
|
74
|
+
// 工作目录可写 + /tmp 可写 + 断网
|
|
75
|
+
base.push('--bind', ctx.cwd, ctx.cwd, '--unshare-net');
|
|
76
|
+
} else {
|
|
77
|
+
// readonly:工作目录也只读
|
|
78
|
+
base.push('--ro-bind', ctx.cwd, ctx.cwd);
|
|
79
|
+
}
|
|
80
|
+
base.push('--chdir', ctx.cwd, '--', '/bin/bash', '-lc', command);
|
|
81
|
+
spawnCmd = 'bwrap';
|
|
82
|
+
spawnArgs = base;
|
|
83
|
+
sandbox = mode;
|
|
84
|
+
} else if (mode !== 'off') {
|
|
85
|
+
note = `沙箱模式 "${mode}" 不可用(需要 Linux + bubblewrap),已降级为直接执行。`;
|
|
86
|
+
sandbox = 'off';
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
return new Promise((resolve) => {
|
|
90
|
+
const child = spawn(spawnCmd, spawnArgs, {
|
|
91
|
+
cwd: ctx.cwd,
|
|
92
|
+
env: buildChildEnv(ctx, ctx?.cfg?.bashEnvFilter !== false), // 默认过滤敏感变量(评估 P2-3,与沙箱档位解耦)
|
|
93
|
+
stdio: ['ignore', 'pipe', 'pipe'],
|
|
94
|
+
detached: true, // 自成进程组:超时/结束可整组清理,孙进程不成孤儿
|
|
95
|
+
});
|
|
96
|
+
let out = '';
|
|
97
|
+
let err = '';
|
|
98
|
+
let done = false;
|
|
99
|
+
let timedOut = false;
|
|
100
|
+
// 输出增量截断:超长输出只保留尾部,避免内存无限累积
|
|
101
|
+
const cap = (s, d) => {
|
|
102
|
+
const t = s + d;
|
|
103
|
+
return t.length > MAX_OUTPUT * 2 ? t.slice(-MAX_OUTPUT * 2) : t;
|
|
104
|
+
};
|
|
105
|
+
const killGroup = (sig) => {
|
|
106
|
+
try {
|
|
107
|
+
process.kill(-child.pid, sig);
|
|
108
|
+
} catch {
|
|
109
|
+
try {
|
|
110
|
+
child.kill(sig);
|
|
111
|
+
} catch {}
|
|
112
|
+
}
|
|
113
|
+
};
|
|
114
|
+
const timer = setTimeout(() => {
|
|
115
|
+
timedOut = true;
|
|
116
|
+
killGroup('SIGKILL');
|
|
117
|
+
}, timeoutSec * 1000);
|
|
118
|
+
// 兜底:close 可能因孙进程持有管道而延迟——先杀整组再收尾;
|
|
119
|
+
// 只有真正超时(timer 已触发)才标 timedOut,正常完成绝不误标(审计 P1-2)
|
|
120
|
+
const forceTimer = setTimeout(() => {
|
|
121
|
+
if (done) return;
|
|
122
|
+
done = true;
|
|
123
|
+
clearTimeout(timer);
|
|
124
|
+
killGroup('SIGKILL');
|
|
125
|
+
resolve({
|
|
126
|
+
ok: true,
|
|
127
|
+
exitCode: timedOut ? 124 : 0,
|
|
128
|
+
timedOut,
|
|
129
|
+
sandbox,
|
|
130
|
+
note: timedOut ? '命令超时,已强杀进程组' : '输出管道未释放,已清理子进程组',
|
|
131
|
+
stdout: tail(out, MAX_OUTPUT),
|
|
132
|
+
stderr: tail(err, MAX_OUTPUT),
|
|
133
|
+
});
|
|
134
|
+
}, timeoutSec * 1000 + 3000);
|
|
135
|
+
|
|
136
|
+
child.stdout.on('data', (d) => {
|
|
137
|
+
out = cap(out, d);
|
|
138
|
+
});
|
|
139
|
+
child.stderr.on('data', (d) => {
|
|
140
|
+
err = cap(err, d);
|
|
141
|
+
});
|
|
142
|
+
child.on('error', (e) => {
|
|
143
|
+
if (done) return;
|
|
144
|
+
done = true;
|
|
145
|
+
clearTimeout(timer);
|
|
146
|
+
clearTimeout(forceTimer);
|
|
147
|
+
resolve({ ok: false, error: `无法启动进程:${e.message}`, sandbox });
|
|
148
|
+
});
|
|
149
|
+
child.on('close', (code) => {
|
|
150
|
+
if (done) return;
|
|
151
|
+
done = true;
|
|
152
|
+
clearTimeout(timer);
|
|
153
|
+
clearTimeout(forceTimer);
|
|
154
|
+
resolve({
|
|
155
|
+
ok: true,
|
|
156
|
+
exitCode: code,
|
|
157
|
+
timedOut,
|
|
158
|
+
sandbox,
|
|
159
|
+
note: note || undefined,
|
|
160
|
+
stdout: tail(out, MAX_OUTPUT),
|
|
161
|
+
stderr: tail(err, MAX_OUTPUT),
|
|
162
|
+
});
|
|
163
|
+
});
|
|
164
|
+
});
|
|
165
|
+
}
|