ppxans-harness 2.4.0 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -32
- package/config/ppx.json +133 -151
- package/package.json +2 -5
- package/src/agent/index.js +3 -3
- package/src/agent/prompts.js +1 -1
- package/src/channels/http.js +76 -17
- package/src/config/providers.js +5 -11
- package/src/llm/client.js +187 -446
- package/src/llm/fence.js +8 -61
- package/src/llm/index.js +1 -1
- package/src/llm/router.js +5 -13
- package/src/mcp/admin.js +292 -0
- package/src/mcp/client.js +107 -5
- package/src/mcp/http.js +217 -0
- package/src/mcp/server.js +387 -0
- package/src/mcp/tasks.js +138 -0
- package/src/plugin/builtin.js +4 -1
package/src/llm/client.js
CHANGED
|
@@ -1,446 +1,187 @@
|
|
|
1
|
-
// src/llm/client.js - LLM 客户端
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
import
|
|
8
|
-
import
|
|
9
|
-
import {
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
//
|
|
119
|
-
async
|
|
120
|
-
this.
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
if (!
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
const
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
toolCalls = null;
|
|
189
|
-
}
|
|
190
|
-
// 纯文本工具调用修复 (吸收 OpenClaw tool-call-repair):
|
|
191
|
-
// 部分模型(本地/DSML)返回文本工具意图而非原生 tool_calls, 从文本恢复
|
|
192
|
-
if (tools.length && (!toolCalls || !toolCalls.length) && typeof content === "string" && content) {
|
|
193
|
-
const parsed = parseToolCalls(content);
|
|
194
|
-
if (parsed.calls.length) {
|
|
195
|
-
toolCalls = parsed.calls.map((c) => ({
|
|
196
|
-
id: c.id,
|
|
197
|
-
type: "function",
|
|
198
|
-
function: { name: c.function.name, arguments: c.function.arguments },
|
|
199
|
-
}));
|
|
200
|
-
content = parsed.clean || null;
|
|
201
|
-
}
|
|
202
|
-
}
|
|
203
|
-
return {
|
|
204
|
-
message: {
|
|
205
|
-
role: message.role || "assistant",
|
|
206
|
-
content,
|
|
207
|
-
tool_calls: toolCalls,
|
|
208
|
-
},
|
|
209
|
-
usage: data?.usage,
|
|
210
|
-
};
|
|
211
|
-
}
|
|
212
|
-
|
|
213
|
-
// 围栏代理循环: 把工具清单注入, 引擎以纯 LLM 输出工具意图, PPX 执行 [P0#1]
|
|
214
|
-
// 围栏上下文 token 预算 (v0.6.6 优化: 替代固定 800/400 字符截断)
|
|
215
|
-
// 按 token 预算动态分配: persona 优先 40%, 历史按"最新优先 + 信息量"分配剩余 60%
|
|
216
|
-
static get FENCE_CTX_TOKEN_BUDGET() { return 2400; }
|
|
217
|
-
static get FENCE_PERSONA_RATIO() { return 0.4; }
|
|
218
|
-
static get FENCE_HIST_BUDGET() { return 2400 * 0.6; }
|
|
219
|
-
// 信息量启发: 含指令/数字/路径/结论的轮次更该保留
|
|
220
|
-
static _histPriority(m) {
|
|
221
|
-
const s = String(m?.content || "");
|
|
222
|
-
let p = 0;
|
|
223
|
-
if (/[查|算|写|建|改|创建|删除|修复|总结|分析|配置|执行|运行|启动|停止|提交|部署|安装|生成|编译|测试]/.test(s)) p += 2;
|
|
224
|
-
if (/[0-9]{2,}/.test(s)) p += 1;
|
|
225
|
-
if (/\.(js|py|md|json|txt|ts|go)\b|[:\\\/][A-Za-z]/.test(s)) p += 2;
|
|
226
|
-
if (/失败|错误|报错|异常|成功|完成|结果|结论|决定|方案/.test(s)) p += 2;
|
|
227
|
-
if (/^(你好|hi|hello|在吗|谢谢|好的|嗯|是的|对|收到|再见)/i.test(s.trim())) p -= 3;
|
|
228
|
-
return p;
|
|
229
|
-
}
|
|
230
|
-
async _proxyChat(messages, { tools, toolRunner, engine }) {
|
|
231
|
-
const fencePrompt = buildFencePrompt(tools);
|
|
232
|
-
// 保留 system/persona 设定 + 最近历史, 避免外部引擎只见"当前问题" [复审 P2]
|
|
233
|
-
const systemMsg = messages.find((m) => m && m.role === "system");
|
|
234
|
-
const nonSys = messages.filter((m) => m && m.role !== "system");
|
|
235
|
-
// person 预算: 固定 40%, 截断到预算内
|
|
236
|
-
const persona = systemMsg?.content
|
|
237
|
-
? "角色设定:\n" + truncateByTokens(systemMsg.content, LLMClient.FENCE_CTX_TOKEN_BUDGET * LLMClient.FENCE_PERSONA_RATIO)
|
|
238
|
-
: null;
|
|
239
|
-
// 历史预算: 剩余 60% 按"最新优先 + 信息量"分配
|
|
240
|
-
const histBudget = LLMClient.FENCE_HIST_BUDGET;
|
|
241
|
-
let histLines = [], used = 0;
|
|
242
|
-
// 1) 最新 6 条按时间倒序, 优先保留(信息量高或最新)
|
|
243
|
-
const recent = nonSys.slice(-6);
|
|
244
|
-
for (let i = recent.length - 1; i >= 0; i--) {
|
|
245
|
-
const m = recent[i];
|
|
246
|
-
const line = (m.role === "user" ? "用户" : "助手") + ": " + String(m.content || "");
|
|
247
|
-
const t = estimateTokens(line);
|
|
248
|
-
if (used + t > histBudget) continue; // 超预算跳过(不截断硬塞)
|
|
249
|
-
histLines.push(line); used += t;
|
|
250
|
-
}
|
|
251
|
-
// 2) 若预算还有余量, 补更早的高信息量轮次
|
|
252
|
-
const older = nonSys.slice(0, Math.max(0, nonSys.length - 6));
|
|
253
|
-
for (let i = older.length - 1; i >= 0; i--) {
|
|
254
|
-
const m = older[i];
|
|
255
|
-
if (LLMClient._histPriority(m) < 2) continue; // 只补高信息量
|
|
256
|
-
const line = (m.role === "user" ? "用户" : "助手") + ": " + String(m.content || "");
|
|
257
|
-
const t = estimateTokens(line);
|
|
258
|
-
if (used + t > histBudget) break;
|
|
259
|
-
histLines.push(line); used += t;
|
|
260
|
-
}
|
|
261
|
-
histLines.reverse(); // 恢复时间正序
|
|
262
|
-
const ctx = [persona, histLines.length ? histLines.join("\n") : null].filter(Boolean).join("\n\n");
|
|
263
|
-
const combined = fencePrompt + "\n\n[上下文]\n" + ctx;
|
|
264
|
-
const finalText = await proxyToolLoop(
|
|
265
|
-
async (context) => {
|
|
266
|
-
// 每条消息: 围栏协议 + 任务 + 累积工具结果
|
|
267
|
-
const msg = context ? combined + "\n\n" + context : combined;
|
|
268
|
-
if (engine === "openclaw") {
|
|
269
|
-
const r = await this._openclawChatAsync([{ role: "user", content: msg }]);
|
|
270
|
-
return r.content;
|
|
271
|
-
}
|
|
272
|
-
const r = await this._dshChatAsync([{ role: "user", content: msg }]);
|
|
273
|
-
return r.content;
|
|
274
|
-
},
|
|
275
|
-
toolRunner,
|
|
276
|
-
{ maxRounds: 8 }
|
|
277
|
-
);
|
|
278
|
-
return { message: { role: "assistant", content: finalText, tool_calls: null }, usage: null };
|
|
279
|
-
}
|
|
280
|
-
|
|
281
|
-
// 辅助 LLM 调用短超时 (毫秒): 压缩/提炼/扩展/经验 等非主对话调用, 快速失败降级, 避免 120s 卡死
|
|
282
|
-
// 使用场景: 模型未运行/网络不通时, 主对话靠 localIntent 或回退, 辅助调用不应阻塞主流程
|
|
283
|
-
static get AUX_TIMEOUT_MS() { return 10000; }
|
|
284
|
-
|
|
285
|
-
async _request(path, jsonBody, { timeoutMs, retryMax } = {}) {
|
|
286
|
-
if (!this.apiKey) throw new Error(`[皮皮虾] LLM 缺少 API key (env=${this.apiKeyEnvName || "?"})`);
|
|
287
|
-
const url = `${this.baseUrl}${path}`;
|
|
288
|
-
const ms = timeoutMs || this.timeoutMs;
|
|
289
|
-
// 辅助调用 (压缩/提炼/扩展等) 传 retryMax:0 禁重试: 短超时 + 不重试 = 快速失败降级, 不阻塞主流程
|
|
290
|
-
const maxRetries = retryMax === undefined ? this.retryMax : retryMax;
|
|
291
|
-
const doFetch = async () => {
|
|
292
|
-
const ctrl = new AbortController();
|
|
293
|
-
const timer = setTimeout(() => ctrl.abort(), ms);
|
|
294
|
-
try {
|
|
295
|
-
const resp = await fetch(url, {
|
|
296
|
-
method: "POST",
|
|
297
|
-
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${this.apiKey}` },
|
|
298
|
-
body: JSON.stringify(jsonBody),
|
|
299
|
-
signal: ctrl.signal,
|
|
300
|
-
});
|
|
301
|
-
if (!resp.ok) {
|
|
302
|
-
const text = await resp.text().catch(() => "");
|
|
303
|
-
const e = new Error(`LLM HTTP ${resp.status}: ${text.slice(0, 300)}`);
|
|
304
|
-
e.status = resp.status; // 结构化状态码, 供 retry.isTransientError 分类
|
|
305
|
-
throw e;
|
|
306
|
-
}
|
|
307
|
-
return await resp.json();
|
|
308
|
-
} finally {
|
|
309
|
-
clearTimeout(timer);
|
|
310
|
-
}
|
|
311
|
-
};
|
|
312
|
-
// 瞬态错误(429/5xx/timeout)单次调用内重试, 非瞬态(400/401/403)立即抛给上层 provider 回退
|
|
313
|
-
return withRetry(doFetch, { maxRetries });
|
|
314
|
-
}
|
|
315
|
-
|
|
316
|
-
// 是否支持逐字流式: 仅直连 HTTP API 后端 (openclaw/deepseek 为外部进程, 一次性返回) [复审 P2]
|
|
317
|
-
get supportsStream() { return this.backend !== "openclaw" && this.backend !== "deepseek"; }
|
|
318
|
-
|
|
319
|
-
// 是否支持原生 tool_calls: 仅 http 后端 (OpenAI 兼容 API)。
|
|
320
|
-
// openclaw 是完整 agent 运行时, 拒绝 PPX 的围栏协议(视为伪协议); dsh 走文本围栏。
|
|
321
|
-
// 工具类任务应优先路由到支持原生 tool_calls 的后端 (实测 LM Studio 原生 tool_calls 全链路通过)。
|
|
322
|
-
get supportsNativeToolCalls() { return this.backend === "http"; }
|
|
323
|
-
|
|
324
|
-
// openclaw 后端: 启动前校验 Node 版本, 不满足则抛中文引导错误 (而非原始报错)
|
|
325
|
-
_openclawReadyOrThrow() {
|
|
326
|
-
if (!nodeVersionOk(process.versions.node)) {
|
|
327
|
-
throw new Error("[皮皮虾] 当前 Node v" + process.versions.node + " 不满足 OpenClaw 引擎要求。\n请升级 Node 至 >=22.22.3 (推荐 26.x);注意 Node 23 与 24.0-24.14 不支持。\n或改用 http 后端配置 API key 直连。");
|
|
328
|
-
}
|
|
329
|
-
if (!this.mjs || !fs.existsSync(this.mjs)) {
|
|
330
|
-
throw new Error("[皮皮虾] OpenClaw 引擎未就绪: mjs 路径不存在 (" + this.mjs + ")。\n请设置环境变量 PPX_OPENCLAW_MJS 指向 openclaw.mjs,或在 config/ppx.json 的 openclaw provider 里填 mjs 字段。");
|
|
331
|
-
}
|
|
332
|
-
}
|
|
333
|
-
|
|
334
|
-
// 翻译 openclaw CLI 的版本类报错为中文引导
|
|
335
|
-
_translateOpenclawError(e) {
|
|
336
|
-
const msg = String((e && e.message) || e);
|
|
337
|
-
if (/Node\.js >= \d+\.\d+\.\d+/.test(msg) || /engines|不满足引擎要求/.test(msg)) {
|
|
338
|
-
return new Error("[皮皮虾] OpenClaw 引擎要求更高的 Node 版本,请升级 Node 至 >=22.22.3 (推荐 26.x) 后重试。原始信息: " + msg.slice(0, 200));
|
|
339
|
-
}
|
|
340
|
-
return e;
|
|
341
|
-
}
|
|
342
|
-
// ===== Provider 健康探测 (Harness 化: 并发探测可用性, 支持快速失败) =====
|
|
343
|
-
// openclaw 后端: 校验本地 Node 版本是否满足引擎要求 (>=22.22.3 / >=24.15 / >=25.9)
|
|
344
|
-
// http 后端: 快速探测 /models (3s 超时), 不发完整请求
|
|
345
|
-
async health() {
|
|
346
|
-
if (this.backend === "openclaw") {
|
|
347
|
-
const okNode = nodeVersionOk(process.versions.node);
|
|
348
|
-
const okMjs = !!this.mjs && fs.existsSync(this.mjs);
|
|
349
|
-
const ok = okNode && okMjs;
|
|
350
|
-
if (!okNode) info("[health] openclaw 不可用: Node v" + process.versions.node + " 不满足引擎要求");
|
|
351
|
-
if (okNode && !okMjs) info("[health] openclaw 不可用: mjs 路径不存在 (" + this.mjs + "),请设置 PPX_OPENCLAW_MJS");
|
|
352
|
-
return ok;
|
|
353
|
-
}
|
|
354
|
-
if (this.backend === "deepseek") {
|
|
355
|
-
const r = this._dshResolveBin();
|
|
356
|
-
if (!r) {
|
|
357
|
-
info("[health] deepseek 不可用: dsh 未就绪 " + this.dshRoot + " (缺 lib/bin.js 或 apps/cli/src/bin.ts)");
|
|
358
|
-
return false;
|
|
359
|
-
}
|
|
360
|
-
if (r.kind === "src" && !fs.existsSync(path.join(this.dshRoot, "node_modules", "tsx"))) {
|
|
361
|
-
info("[health] deepseek 不可用: dsh 源码形态依赖未安装 (请运行 npm run dsh:install 或设 PPX_DSH_ROOT)");
|
|
362
|
-
return false;
|
|
363
|
-
}
|
|
364
|
-
return nodeVersionOk(process.versions.node);
|
|
365
|
-
}
|
|
366
|
-
if (!this.apiKey) return false;
|
|
367
|
-
try {
|
|
368
|
-
const ctrl = new AbortController();
|
|
369
|
-
const t = setTimeout(() => ctrl.abort(), 3000);
|
|
370
|
-
const r = await fetch(this.baseUrl + "/models", {
|
|
371
|
-
headers: { "Authorization": "Bearer " + this.apiKey },
|
|
372
|
-
signal: ctrl.signal,
|
|
373
|
-
});
|
|
374
|
-
clearTimeout(t);
|
|
375
|
-
return r.ok;
|
|
376
|
-
} catch (e) {
|
|
377
|
-
warn("[health] " + this.providerId + " 探测失败:", e.message);
|
|
378
|
-
return false;
|
|
379
|
-
}
|
|
380
|
-
}
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
// 流式 chat: 逐块回调 (SSE), 返回累积文本
|
|
384
|
-
// onDelta(content) 每次增量, onDone(full) 结束
|
|
385
|
-
async streamChat(messages, { temperature = 0.7, maxTokens = 4096, onDelta, signal } = {}) {
|
|
386
|
-
if (this.backend === "openclaw") {
|
|
387
|
-
// OpenClaw CLI 非流式: 一次性返回全文 (先可用, 后续可切 SSE)
|
|
388
|
-
const r = await this._openclawChatAsync(messages);
|
|
389
|
-
if (onDelta) onDelta(r.content);
|
|
390
|
-
return r.content;
|
|
391
|
-
}
|
|
392
|
-
if (this.backend === "deepseek") {
|
|
393
|
-
// dsh headless 非流式: 一次性返回全文
|
|
394
|
-
const r = await this._dshChatAsync(messages);
|
|
395
|
-
if (onDelta) onDelta(r.content);
|
|
396
|
-
return r.content;
|
|
397
|
-
}
|
|
398
|
-
if (!this.apiKey) throw new Error(`[皮皮虾] LLM 缺少 API key`);
|
|
399
|
-
const url = `${this.baseUrl}/chat/completions`;
|
|
400
|
-
const ctrl = new AbortController();
|
|
401
|
-
const timer = setTimeout(() => ctrl.abort(), this.timeoutMs);
|
|
402
|
-
const extSig = signal || null;
|
|
403
|
-
if (extSig) extSig.addEventListener("abort", () => ctrl.abort());
|
|
404
|
-
let full = "";
|
|
405
|
-
try {
|
|
406
|
-
const resp = await fetch(url, {
|
|
407
|
-
method: "POST",
|
|
408
|
-
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${this.apiKey}` },
|
|
409
|
-
body: JSON.stringify({ model: this.model, messages, temperature, max_tokens: maxTokens, stream: true }),
|
|
410
|
-
signal: ctrl.signal,
|
|
411
|
-
});
|
|
412
|
-
if (!resp.ok) {
|
|
413
|
-
const txt = await resp.text().catch(() => "");
|
|
414
|
-
throw new Error(`LLM HTTP ${resp.status}: ${txt.slice(0, 300)}`);
|
|
415
|
-
}
|
|
416
|
-
if (!resp.body) { throw new Error("响应无 body, 不支持流式"); }
|
|
417
|
-
const reader = resp.body.getReader();
|
|
418
|
-
const decoder = new TextDecoder();
|
|
419
|
-
let buf = "";
|
|
420
|
-
let streamDone = false; // 独立结束信号, 不污染 reader.read() 的 done
|
|
421
|
-
while (!streamDone) {
|
|
422
|
-
const { done, value } = await reader.read();
|
|
423
|
-
if (done) break;
|
|
424
|
-
buf += decoder.decode(value, { stream: true });
|
|
425
|
-
// 按行解析 SSE
|
|
426
|
-
let idx;
|
|
427
|
-
while (!streamDone && (idx = buf.indexOf("\n")) !== -1) {
|
|
428
|
-
const line = buf.slice(0, idx).trim().replace(/\r$/, "");
|
|
429
|
-
buf = buf.slice(idx + 1);
|
|
430
|
-
if (!line.startsWith("data:")) continue;
|
|
431
|
-
const data = line.slice(5).trim();
|
|
432
|
-
// 官方结束信号: 仅匹配空格式的 "data: [DONE]", 不做 buf 全文搜防误截
|
|
433
|
-
if (data === "[DONE]") { streamDone = true; break; }
|
|
434
|
-
try {
|
|
435
|
-
const j = JSON.parse(data);
|
|
436
|
-
const delta = j.choices?.[0]?.delta?.content;
|
|
437
|
-
if (delta) { full += delta; onDelta && onDelta(delta); }
|
|
438
|
-
} catch {}
|
|
439
|
-
}
|
|
440
|
-
}
|
|
441
|
-
return full;
|
|
442
|
-
} finally {
|
|
443
|
-
clearTimeout(timer);
|
|
444
|
-
}
|
|
445
|
-
}
|
|
446
|
-
}
|
|
1
|
+
// src/llm/client.js - LLM 客户端 (自研底座: 仅 OpenAI 兼容 HTTP 直连)
|
|
2
|
+
// 后端模式: backend="http" (默认唯一后端, 零依赖, 用 fetch)
|
|
3
|
+
// - 直连任意 OpenAI 兼容 API: OpenAI/DeepSeek/火山/通义/智谱/本地 (lmstudio/ollama/vLLM)
|
|
4
|
+
// - 原生 tool_calls + 文本工具调用修复 (围栏/DSML 解析, 自研)
|
|
5
|
+
// - SSE 流式 / 瞬态错误重试 / provider 健康探测
|
|
6
|
+
// 历史: v2.4.0 前支持 openclaw/dsh 外部引擎底座, v2.5.0 起全部移除, 只保留自研 http 底座。
|
|
7
|
+
import { parseToolCalls } from "./fence.js";
|
|
8
|
+
import { withRetry } from "./retry.js";
|
|
9
|
+
import { warn } from "../utils/logger.js";
|
|
10
|
+
|
|
11
|
+
export class LLMClient {
|
|
12
|
+
constructor(provider) {
|
|
13
|
+
this.providerId = provider.id || "http";
|
|
14
|
+
// 唯一后端: http (OpenAI 兼容 API 直连)
|
|
15
|
+
this.backend = "http";
|
|
16
|
+
this.baseUrl = (provider.base_url || "").replace(/\/$/, "");
|
|
17
|
+
this.apiKey = provider.api_key || process.env[provider.api_key_env] || "";
|
|
18
|
+
this.apiKeyEnvName = provider.api_key_env || ""; // 供缺失 key 报错时提示应设置的环境变量名
|
|
19
|
+
this.model = provider.model || provider.models?.chat || "gpt-4o-mini";
|
|
20
|
+
this.vision = !!provider.vision; // 是否支持多模态 (视觉) — 标记后才会注入图片到 user 消息
|
|
21
|
+
// 上下文窗口 (token): 供 agent 据此收紧会话历史预算, 防止本地小模型溢出。
|
|
22
|
+
// 可选字段, provider 未配置时用保守默认 8192 (绝不因未知窗口放大历史)。
|
|
23
|
+
this.context_window = Number(provider.context_window) || Number(provider.models?.context_window) || 8192;
|
|
24
|
+
this.timeoutMs = provider.timeout_ms || 120000;
|
|
25
|
+
this.retryMax = provider.retry_max ?? 3; // 单次调用内瞬态错误重试次数 (429/5xx/timeout)
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// 原生 chat (无工具)
|
|
29
|
+
// timeoutMs/retryMax: 可选覆盖 provider 默认 (辅助调用传短超时+禁重试快速失败, 见 AUX_TIMEOUT_MS)
|
|
30
|
+
async chat(messages, { temperature = 0.7, maxTokens = 2048, timeoutMs, retryMax } = {}) {
|
|
31
|
+
const data = await this._request("/chat/completions", { model: this.model, messages, temperature, max_tokens: maxTokens }, { timeoutMs, retryMax });
|
|
32
|
+
const m1 = data?.choices?.[0]?.message;
|
|
33
|
+
let content = m1?.content;
|
|
34
|
+
// 本地推理模型兜底: thinking 吃满 token 时 content 为空, 用 reasoning_content 降级, 避免误判"断线/失败"并写入污染记忆
|
|
35
|
+
if (!content && m1?.reasoning_content) content = "[思考] " + m1.reasoning_content;
|
|
36
|
+
if (!content) throw new Error("LLM 返回空内容");
|
|
37
|
+
return { content, usage: data?.usage };
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
// API chat (支持工具调用), 返回完整 message (含 tool_calls)
|
|
41
|
+
async apiChat(messages, { tools = [], temperature = 0.7, maxTokens = 4096, toolRunner = null, timeoutMs, retryMax } = {}) {
|
|
42
|
+
const body = { model: this.model, messages, temperature, max_tokens: maxTokens };
|
|
43
|
+
if (tools.length) body.tools = tools;
|
|
44
|
+
const data = await this._request("/chat/completions", body, { timeoutMs, retryMax });
|
|
45
|
+
const message = data?.choices?.[0]?.message;
|
|
46
|
+
if (!message) throw new Error("LLM 返回空 message");
|
|
47
|
+
let toolCalls = message.tool_calls || null;
|
|
48
|
+
let content = message.content || null;
|
|
49
|
+
// 本地推理模型兜底: 无正文且无工具调用时, 用 reasoning_content 降级(避免误判断线/失败并写入污染记忆)
|
|
50
|
+
if (!content && !toolCalls?.length && message.reasoning_content) {
|
|
51
|
+
content = "[思考] " + message.reasoning_content;
|
|
52
|
+
toolCalls = null;
|
|
53
|
+
}
|
|
54
|
+
// 纯文本工具调用修复 (自研围栏 ⟪tool⟫ / DSML 解析):
|
|
55
|
+
// 部分模型(本地/DSML)返回文本工具意图而非原生 tool_calls, 从文本恢复
|
|
56
|
+
if (tools.length && (!toolCalls || !toolCalls.length) && typeof content === "string" && content) {
|
|
57
|
+
const parsed = parseToolCalls(content);
|
|
58
|
+
if (parsed.calls.length) {
|
|
59
|
+
toolCalls = parsed.calls.map((c) => ({
|
|
60
|
+
id: c.id,
|
|
61
|
+
type: "function",
|
|
62
|
+
function: { name: c.function.name, arguments: c.function.arguments },
|
|
63
|
+
}));
|
|
64
|
+
content = parsed.clean || null;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
return {
|
|
68
|
+
message: {
|
|
69
|
+
role: message.role || "assistant",
|
|
70
|
+
content,
|
|
71
|
+
tool_calls: toolCalls,
|
|
72
|
+
},
|
|
73
|
+
usage: data?.usage,
|
|
74
|
+
};
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// 辅助 LLM 调用短超时 (毫秒): 压缩/提炼/扩展/经验 等非主对话调用, 快速失败降级, 避免 120s 卡死
|
|
78
|
+
// 使用场景: 模型未运行/网络不通时, 主对话靠 localIntent 或回退, 辅助调用不应阻塞主流程
|
|
79
|
+
static get AUX_TIMEOUT_MS() { return 10000; }
|
|
80
|
+
|
|
81
|
+
async _request(path, jsonBody, { timeoutMs, retryMax } = {}) {
|
|
82
|
+
if (!this.apiKey) throw new Error(`[皮皮虾] LLM 缺少 API key (env=${this.apiKeyEnvName || "?"})`);
|
|
83
|
+
const url = `${this.baseUrl}${path}`;
|
|
84
|
+
const ms = timeoutMs || this.timeoutMs;
|
|
85
|
+
// 辅助调用 (压缩/提炼/扩展等) 传 retryMax:0 禁重试: 短超时 + 不重试 = 快速失败降级, 不阻塞主流程
|
|
86
|
+
const maxRetries = retryMax === undefined ? this.retryMax : retryMax;
|
|
87
|
+
const doFetch = async () => {
|
|
88
|
+
const ctrl = new AbortController();
|
|
89
|
+
const timer = setTimeout(() => ctrl.abort(), ms);
|
|
90
|
+
try {
|
|
91
|
+
const resp = await fetch(url, {
|
|
92
|
+
method: "POST",
|
|
93
|
+
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${this.apiKey}` },
|
|
94
|
+
body: JSON.stringify(jsonBody),
|
|
95
|
+
signal: ctrl.signal,
|
|
96
|
+
});
|
|
97
|
+
if (!resp.ok) {
|
|
98
|
+
const text = await resp.text().catch(() => "");
|
|
99
|
+
const e = new Error(`LLM HTTP ${resp.status}: ${text.slice(0, 300)}`);
|
|
100
|
+
e.status = resp.status; // 结构化状态码, 供 retry.isTransientError 分类
|
|
101
|
+
throw e;
|
|
102
|
+
}
|
|
103
|
+
return await resp.json();
|
|
104
|
+
} finally {
|
|
105
|
+
clearTimeout(timer);
|
|
106
|
+
}
|
|
107
|
+
};
|
|
108
|
+
// 瞬态错误(429/5xx/timeout)单次调用内重试, 非瞬态(400/401/403)立即抛给上层 provider 回退
|
|
109
|
+
return withRetry(doFetch, { maxRetries });
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
// 自研 http 底座: 支持逐字流式 (SSE)
|
|
113
|
+
get supportsStream() { return true; }
|
|
114
|
+
|
|
115
|
+
// 自研 http 底座: 支持原生 tool_calls (OpenAI 兼容 API)
|
|
116
|
+
get supportsNativeToolCalls() { return true; }
|
|
117
|
+
|
|
118
|
+
// Provider 健康探测: 快速探测 /models (3s 超时), 不发完整请求
|
|
119
|
+
async health() {
|
|
120
|
+
if (!this.apiKey) return false;
|
|
121
|
+
try {
|
|
122
|
+
const ctrl = new AbortController();
|
|
123
|
+
const t = setTimeout(() => ctrl.abort(), 3000);
|
|
124
|
+
const r = await fetch(this.baseUrl + "/models", {
|
|
125
|
+
headers: { "Authorization": "Bearer " + this.apiKey },
|
|
126
|
+
signal: ctrl.signal,
|
|
127
|
+
});
|
|
128
|
+
clearTimeout(t);
|
|
129
|
+
return r.ok;
|
|
130
|
+
} catch (e) {
|
|
131
|
+
warn("[health] " + this.providerId + " 探测失败:", e.message);
|
|
132
|
+
return false;
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// 流式 chat: 逐块回调 (SSE), 返回累积文本
|
|
137
|
+
// onDelta(content) 每次增量, onDone(full) 结束
|
|
138
|
+
async streamChat(messages, { temperature = 0.7, maxTokens = 4096, onDelta, signal } = {}) {
|
|
139
|
+
if (!this.apiKey) throw new Error(`[皮皮虾] LLM 缺少 API key`);
|
|
140
|
+
const url = `${this.baseUrl}/chat/completions`;
|
|
141
|
+
const ctrl = new AbortController();
|
|
142
|
+
const timer = setTimeout(() => ctrl.abort(), this.timeoutMs);
|
|
143
|
+
const extSig = signal || null;
|
|
144
|
+
if (extSig) extSig.addEventListener("abort", () => ctrl.abort());
|
|
145
|
+
let full = "";
|
|
146
|
+
try {
|
|
147
|
+
const resp = await fetch(url, {
|
|
148
|
+
method: "POST",
|
|
149
|
+
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${this.apiKey}` },
|
|
150
|
+
body: JSON.stringify({ model: this.model, messages, temperature, max_tokens: maxTokens, stream: true }),
|
|
151
|
+
signal: ctrl.signal,
|
|
152
|
+
});
|
|
153
|
+
if (!resp.ok) {
|
|
154
|
+
const txt = await resp.text().catch(() => "");
|
|
155
|
+
throw new Error(`LLM HTTP ${resp.status}: ${txt.slice(0, 300)}`);
|
|
156
|
+
}
|
|
157
|
+
if (!resp.body) { throw new Error("响应无 body, 不支持流式"); }
|
|
158
|
+
const reader = resp.body.getReader();
|
|
159
|
+
const decoder = new TextDecoder();
|
|
160
|
+
let buf = "";
|
|
161
|
+
let streamDone = false; // 独立结束信号, 不污染 reader.read() 的 done
|
|
162
|
+
while (!streamDone) {
|
|
163
|
+
const { done, value } = await reader.read();
|
|
164
|
+
if (done) break;
|
|
165
|
+
buf += decoder.decode(value, { stream: true });
|
|
166
|
+
// 按行解析 SSE
|
|
167
|
+
let idx;
|
|
168
|
+
while (!streamDone && (idx = buf.indexOf("\n")) !== -1) {
|
|
169
|
+
const line = buf.slice(0, idx).trim().replace(/\r$/, "");
|
|
170
|
+
buf = buf.slice(idx + 1);
|
|
171
|
+
if (!line.startsWith("data:")) continue;
|
|
172
|
+
const data = line.slice(5).trim();
|
|
173
|
+
// 官方结束信号: 仅匹配空格式的 "data: [DONE]", 不做 buf 全文搜防误截
|
|
174
|
+
if (data === "[DONE]") { streamDone = true; break; }
|
|
175
|
+
try {
|
|
176
|
+
const j = JSON.parse(data);
|
|
177
|
+
const delta = j.choices?.[0]?.delta?.content;
|
|
178
|
+
if (delta) { full += delta; onDelta && onDelta(delta); }
|
|
179
|
+
} catch {}
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
return full;
|
|
183
|
+
} finally {
|
|
184
|
+
clearTimeout(timer);
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
}
|